From 9ae02ac62abb54e4770adc07ff6452fff5928563 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 00:48:33 +0800 Subject: [PATCH 01/20] add independent multi-rail RDMA descriptor reads Build a rail planner and bounded read state machine directly on ContextStore main. Keep each rail's incoming stripes in a compact registered buffer using the existing tag-15 SGE protocol, then validate stripe coverage, acknowledgements, checksums, and post-read object identity before publishing bytes. Support explicit per-node listener routes, weighted byte scheduling, runtime enablement and cooldown, cancellation, resource budgets, per-rail metrics and sysfs topology. Add a labeled benchmark CLI, Mock fault and overhead checks, hardware-gated E2E tests, CI coverage, and compatibility documentation. --- .github/workflows/ci.yml | 31 + Cargo.lock | 1 + README.md | 42 + kv-service/client-rs/Cargo.toml | 8 +- .../client-rs/src/bin/rail_read_bench.rs | 232 +++++ kv-service/client-rs/src/bin/rdma_bench.rs | 11 +- kv-service/client-rs/src/lib.rs | 50 +- kv-service/client-rs/src/rail_read.rs | 925 ++++++++++++++++++ kv-service/client-rs/src/rail_read/tests.rs | 686 +++++++++++++ kv-service/client-rs/src/rdma.rs | 113 ++- .../client-rs/tests/local_cluster_e2e.rs | 15 +- kv-service/client-rs/tests/rail_read_e2e.rs | 202 ++++ 12 files changed, 2292 insertions(+), 24 deletions(-) create mode 100644 kv-service/client-rs/src/bin/rail_read_bench.rs create mode 100644 kv-service/client-rs/src/rail_read.rs create mode 100644 kv-service/client-rs/src/rail_read/tests.rs create mode 100644 kv-service/client-rs/tests/rail_read_e2e.rs diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4f175fa..f499b84 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -89,6 +89,37 @@ jobs: working-directory: kv-service/server run: cargo test --release + # Compile the real Verbs path and run hardware-independent rail state-machine tests. + rust-rdma: + name: Rust RDMA client + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Install stable toolchain + uses: dtolnay/rust-toolchain@stable + with: + components: clippy + + - name: Install Verbs build dependencies + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + protobuf-compiler libclang-dev pkg-config \ + libibverbs-dev librdmacm-dev + + - name: Test RDMA SDK, benchmark CLI, and Mock state machine + working-directory: kv-service/client-rs + run: cargo test --features rdma --lib --bins --test rail_read_e2e + + - name: Check RDMA service feature combination + working-directory: kv-service/server + run: cargo check --features rdma,io-uring,metrics + + - name: Clippy RDMA client + working-directory: kv-service/client-rs + run: cargo clippy --features rdma --all-targets -- -D warnings + # ========================================================================== # E2E: isolated Redis + two loopback KVService nodes # ========================================================================== diff --git a/Cargo.lock b/Cargo.lock index 5caefbb..d063aef 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -355,6 +355,7 @@ dependencies = [ "tokio-stream", "tonic", "tonic-build", + "twox-hash", ] [[package]] diff --git a/README.md b/README.md index c773762..0cd13fb 100644 --- a/README.md +++ b/README.md @@ -305,6 +305,48 @@ Run the Python benchmark wrapper through the root Makefile with: make bench ``` +### Multi-rail descriptor reads (experimental) + +The Rust SDK can read one striped object over multiple local RDMA devices and +listeners on the same owning storage node. This uses the existing descriptor +GET and tag-15 SGE protocol; object placement and disk stripes are unchanged. +`LookupObject` advertises one primary RDMA endpoint per node, so each rail +explicitly maps that advertised endpoint to a listener on the same node. +Different rail entries for one node must use different local device/port pairs +and different remote listeners. A seventh comma-separated rail field may set +a positive relative scheduling weight; omitted weights default to 1. The +reader also reports each local device's sysfs NUMA node and PCI address when +the host exposes them. + +```bash +# Run against an already stored striped object. Set the server's +# CS_RDMA_DEVICES for both listeners and advertise its primary listener. +./target/release/cs-rail-read-bench \ + --environment physical --coordinator http://10.0.0.1:50051 \ + --namespace bench --object-key large-object \ + --rail 'r0,mlx5_0,10.0.0.1:50053,10.0.0.1:50053,1,3' \ + --rail 'r1,mlx5_1,10.0.0.1:50053,10.0.1.1:50054,1,3' \ + --warmup 1 --iterations 5 +``` + +Use only the first `--rail` for a comparable single-rail run. The CLI labels +physical RDMA and Soft-RoCE separately, reports per-rail bytes, and hashes +every returned object. Set the server's cache policy and disk-read forcing +identically for both runs; these flags cannot prove a network bottleneck by +themselves. The SDK's `KvClient::read_multi_rail_into` rechecks descriptor and +placement identity after the transfer and publishes bytes only after all +rails and checksums succeed. A failed or cancelled read leaves its caller +buffer unchanged. Each rail owns a compact registered receive buffer, and +the reader bounds active requests, in-flight bytes, staging, and registered +memory. No in-request transparent retry is attempted. + +For hardware-independent scheduling and failure checks, run +`cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma +rail_read::tests`. The ignored `rail_read_e2e` tests require two reachable +RDMA listeners and the `CS_RAIL_*` endpoint/device environment variables. +The ignored `software_only_mock_benchmark` exercises scheduling and memory +copies; its throughput is **not** an RDMA hardware result. + --- ## Deployment shapes diff --git a/kv-service/client-rs/Cargo.toml b/kv-service/client-rs/Cargo.toml index 8ad21a0..cc3983d 100644 --- a/kv-service/client-rs/Cargo.toml +++ b/kv-service/client-rs/Cargo.toml @@ -14,6 +14,11 @@ name = "cs-rdma-bench" path = "src/bin/rdma_bench.rs" required-features = ["rdma"] +[[bin]] +name = "cs-rail-read-bench" +path = "src/bin/rail_read_bench.rs" +required-features = ["rdma"] + [dependencies] # gRPC (versions aligned with the server) tonic = "0.11" @@ -29,10 +34,11 @@ clap = { version = "4", features = ["derive"] } rdma-sys = { version = "0.3", optional = true } anyhow = { version = "1", optional = true } libc = { version = "0.2", optional = true } +twox-hash = { version = "1.6", optional = true } [features] default = [] -rdma = ["dep:rdma-sys", "dep:anyhow", "dep:libc"] +rdma = ["dep:rdma-sys", "dep:anyhow", "dep:libc", "dep:twox-hash"] [build-dependencies] tonic-build = "0.11" diff --git a/kv-service/client-rs/src/bin/rail_read_bench.rs b/kv-service/client-rs/src/bin/rail_read_bench.rs new file mode 100644 index 0000000..5bf28c7 --- /dev/null +++ b/kv-service/client-rs/src/bin/rail_read_bench.rs @@ -0,0 +1,232 @@ +//! Compare one Worker reading a striped object through one or more RDMA rails. + +use anyhow::{anyhow, Context, Result}; +use clap::Parser; +use contextstore_client_rs::rail_read::{RailLimits, RailReader, RailRoute}; +use contextstore_client_rs::rdma::RdmaClientConfig; +use contextstore_client_rs::KvClient; +use std::sync::Arc; +use std::time::Instant; + +#[derive(Parser)] +#[command(about = "Read one ContextStore object over independently configured RDMA rails")] +struct Args { + #[arg(long)] + coordinator: String, + #[arg(long)] + namespace: String, + #[arg(long)] + object_key: String, + /// Repeat: id,local_device,advertised_endpoint,listener,port,gid_index[,weight]. + #[arg(long = "rail", required = true)] + rails: Vec, + /// State which real Verbs environment produced these measurements. + #[arg(long, value_parser = ["physical", "soft-roce"])] + environment: String, + #[arg(long, default_value_t = 1)] + warmup: usize, + #[arg(long, default_value_t = 5)] + iterations: usize, +} + +fn parse_route(spec: &str) -> Result { + let fields: Vec<_> = spec.split(',').map(str::trim).collect(); + if !(6..=7).contains(&fields.len()) || fields[..4].iter().any(|field| field.is_empty()) { + return Err(anyhow!( + "rail spec must be id,device,advertised_endpoint,listener,port,gid_index[,weight]" + )); + } + let port = fields[4].parse::().context("rail port is not a u8")?; + let gid = fields[5] + .parse::() + .context("rail GID index is not a u8")?; + let weight = fields + .get(6) + .map(|value| value.parse::().context("rail weight is not a u32")) + .transpose()? + .unwrap_or(1); + if weight == 0 { + return Err(anyhow!("rail weight must be positive")); + } + Ok(RailRoute::new( + fields[0], + fields[2], + RdmaClientConfig::new(fields[3], fields[1]) + .with_port(port) + .with_gid_index(gid), + ) + .with_weight(weight)) +} + +#[cfg(target_os = "linux")] +fn process_usage() -> (u64, u64, u64) { + let mut usage = unsafe { std::mem::zeroed::() }; + let status = unsafe { libc::getrusage(libc::RUSAGE_SELF, &mut usage) }; + assert_eq!(status, 0, "getrusage failed"); + let micros = |time: libc::timeval| time.tv_sec as u64 * 1_000_000 + time.tv_usec as u64; + ( + micros(usage.ru_utime), + micros(usage.ru_stime), + usage.ru_maxrss as u64, + ) +} + +#[cfg(not(target_os = "linux"))] +fn process_usage() -> (u64, u64, u64) { + (0, 0, 0) +} + +#[tokio::main] +async fn main() -> Result<()> { + let args = Args::parse(); + if args.iterations == 0 { + return Err(anyhow!("--iterations must be positive")); + } + let routes = args + .rails + .iter() + .map(|spec| parse_route(spec)) + .collect::>>()?; + let reader = Arc::new(RailReader::new(routes.clone(), RailLimits::default())?); + let endpoint = + if args.coordinator.starts_with("http://") || args.coordinator.starts_with("https://") { + args.coordinator.clone() + } else { + format!("http://{}", args.coordinator) + }; + let mut client = KvClient::connect(endpoint) + .await + .map_err(|error| anyhow!(error.to_string()))?; + let lookup = client + .lookup_object(&args.namespace, &args.object_key) + .await? + .ok_or_else(|| anyhow!("object not found"))?; + let size = usize::try_from(lookup.descriptor.size)?; + let mut destination = vec![0u8; size]; + for route in &routes { + println!( + "route,id={},device={},port={},gid={},weight={},advertised={},listener={}", + route.id, + route.connection.device, + route.connection.port, + route.connection.gid_index, + route.weight, + route.advertised_endpoint, + route.connection.endpoint + ); + } + for _ in 0..args.warmup { + client + .read_multi_rail_into( + Arc::clone(&reader), + &args.namespace, + &args.object_key, + &mut destination, + None, + ) + .await? + .ok_or_else(|| anyhow!("object disappeared during warmup"))?; + } + let warmup_bytes: Vec<_> = reader.snapshots().iter().map(|rail| rail.bytes).collect(); + let mut times = Vec::with_capacity(args.iterations); + let mut cpu_user_us = 0u64; + let mut cpu_system_us = 0u64; + for iteration in 0..args.iterations { + destination.fill(0xA5); + let before_cpu = process_usage(); + let started = Instant::now(); + let bytes = client + .read_multi_rail_into( + Arc::clone(&reader), + &args.namespace, + &args.object_key, + &mut destination, + None, + ) + .await? + .ok_or_else(|| anyhow!("object disappeared during benchmark"))?; + let elapsed = started.elapsed(); + let after_cpu = process_usage(); + cpu_user_us += after_cpu.0 - before_cpu.0; + cpu_system_us += after_cpu.1 - before_cpu.1; + if bytes != size { + return Err(anyhow!("short read: {bytes} of {size} bytes")); + } + let checksum = twox_hash::xxh3::hash64(&destination); + let gib_per_s = bytes as f64 / elapsed.as_secs_f64() / 1024f64.powi(3); + println!( + "sample,environment={},rails={},iteration={},bytes={},latency_us={},gib_per_s={gib_per_s:.3},xxh3={checksum:016x}", + args.environment, + routes.len(), + iteration + 1, + bytes, + elapsed.as_micros(), + ); + times.push(elapsed); + } + times.sort_unstable(); + let median = times[times.len() / 2]; + let rail_bytes: Vec<_> = reader + .snapshots() + .iter() + .zip(warmup_bytes) + .map(|(snapshot, warmup)| snapshot.bytes - warmup) + .collect(); + println!( + "summary,environment={},rails={},bytes_per_iter={},iters={},median_us={},cpu_user_us={},cpu_system_us={},peak_rss_kb={},rail_bytes={rail_bytes:?}", + args.environment, + routes.len(), + size, + args.iterations, + median.as_micros(), + cpu_user_us, + cpu_system_us, + process_usage().2 + ); + for snapshot in reader.snapshots() { + let completed = snapshot.reads_ok + snapshot.reads_err; + let average_us = snapshot.duration_us / completed.max(1); + println!( + "rail_stats,id={},device={},listener={},numa={:?},pci={:?},healthy={},cooldown_ms={},reads_ok={},reads_err={},bytes={},avg_us={},inflight_requests={},inflight_bytes={},peak_inflight_bytes={},registered_bytes_reserved={},peak_registered_bytes={}", + snapshot.id, + snapshot.device, + snapshot.listener, + snapshot.topology.numa_node, + snapshot.topology.pci_bdf, + snapshot.healthy, + snapshot.cooldown_ms, + snapshot.reads_ok, + snapshot.reads_err, + snapshot.bytes, + average_us, + snapshot.inflight_requests, + snapshot.inflight_bytes, + snapshot.peak_inflight_bytes, + snapshot.registered_bytes, + snapshot.peak_registered_bytes + ); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rail_spec_keeps_advertised_owner_separate_from_listener() { + let route = parse_route("r1,mlx5_1,10.0.0.1:50053,10.0.1.1:50054,1,3").unwrap(); + assert_eq!(route.id, "r1"); + assert_eq!(route.advertised_endpoint, "10.0.0.1:50053"); + assert_eq!(route.connection.endpoint, "10.0.1.1:50054"); + assert_eq!(route.connection.device, "mlx5_1"); + assert_eq!(route.connection.port, 1); + assert_eq!(route.connection.gid_index, 3); + } + + #[test] + fn rail_spec_accepts_a_capacity_weight() { + let route = parse_route("r1,mlx5_1,10.0.0.1:50053,10.0.1.1:50054,1,3,4").unwrap(); + assert_eq!(route.weight, 4); + } +} diff --git a/kv-service/client-rs/src/bin/rdma_bench.rs b/kv-service/client-rs/src/bin/rdma_bench.rs index 75cc4c6..19454e4 100644 --- a/kv-service/client-rs/src/bin/rdma_bench.rs +++ b/kv-service/client-rs/src/bin/rdma_bench.rs @@ -161,7 +161,9 @@ fn main() -> Result<()> { /// WRITE + commit + striped O_DIRECT pwrite) instead of dedup-skipping. fn run_put(args: &Args) -> Result<()> { if args.coordinator.is_some() { - return Err(anyhow!("--mode put does not support --coordinator (single endpoint only)")); + return Err(anyhow!( + "--mode put does not support --coordinator (single endpoint only)" + )); } let namespace = args .namespace @@ -697,8 +699,7 @@ fn run_multi_endpoint(args: &Args, coordinator: &str) -> Result<()> { let seg_len = (buf_size / sge_segments).max(1) as u64; let mut segments = Vec::with_capacity(sge_segments); let mut off = 0u64; - let (base, rkey, total) = - (view.addr(), view.rkey(), buf_size as u64); + let (base, rkey, total) = (view.addr(), view.rkey(), buf_size as u64); while off < total { let n = seg_len.min(total - off); segments.push((base + off, rkey, n)); @@ -909,6 +910,10 @@ mod tests { #[test] fn readable_selector_encodes_canonical_key() { let args = Args { + mode: "get".to_string(), + put_mb: 480, + ttl_seconds: 0, + sge_segments: 0, key: None, namespace: Some("rust-bench".to_string()), object_key: Some("rdma-checksum0/__combined__".to_string()), diff --git a/kv-service/client-rs/src/lib.rs b/kv-service/client-rs/src/lib.rs index 096266c..b5b0dda 100644 --- a/kv-service/client-rs/src/lib.rs +++ b/kv-service/client-rs/src/lib.rs @@ -19,6 +19,10 @@ pub mod pb { #[cfg(feature = "rdma")] pub mod rdma; +/// Failure-atomic multi-rail descriptor reads over the native RDMA path. +#[cfg(feature = "rdma")] +pub mod rail_read; + use pb::kv_service_client::KvServiceClient; use prost::bytes::Bytes; use tonic::transport::Channel; @@ -219,6 +223,46 @@ impl KvClient { })) } + /// Look up, read via several RDMA rails, and recheck identity before + /// publishing bytes to the caller. A concurrent rewrite fails atomically. + #[cfg(feature = "rdma")] + pub async fn read_multi_rail_into( + &mut self, + reader: std::sync::Arc, + namespace: &str, + object_key: &str, + destination: &mut [u8], + cancel: Option, + ) -> anyhow::Result> { + let initial = match self.lookup_object(namespace, object_key).await? { + Some(lookup) => lookup, + None => return Ok(None), + }; + let size = usize::try_from(initial.descriptor.size)?; + if destination.len() < size { + return Err(rail_read::RailReadError::BufferTooSmall { + need: size, + have: destination.len(), + } + .into()); + } + let placement = initial + .placement + .clone() + .ok_or_else(|| anyhow::anyhow!("lookup returned no RDMA placement"))?; + let descriptor = initial.descriptor.clone(); + let payload = tokio::task::spawn_blocking(move || { + reader.read_staged(&descriptor, &placement, cancel.as_ref()) + }) + .await??; + let current = self + .lookup_object(namespace, object_key) + .await? + .ok_or(rail_read::RailReadError::VersionChanged)?; + let copied = rail_read::commit_if_unchanged(&initial, ¤t, &payload, destination)?; + Ok(Some(copied)) + } + /// Batched [`Self::lookup_object`]: one round trip for N keys. Results /// align with the input order; missing objects yield `None` at their /// position. @@ -625,9 +669,9 @@ impl KvClient { let offset = usize::try_from(chunk.offset).map_err(|_| { tonic::Status::internal(format!("negative chunk offset {}", chunk.offset)) })?; - let end = offset.checked_add(chunk.data.len()).ok_or_else(|| { - tonic::Status::internal("chunk offset + length overflows usize") - })?; + let end = offset + .checked_add(chunk.data.len()) + .ok_or_else(|| tonic::Status::internal("chunk offset + length overflows usize"))?; if end > dst.len() { return Err(tonic::Status::internal(format!( "chunk [{offset}, {end}) exceeds destination buffer of {} bytes", diff --git a/kv-service/client-rs/src/rail_read.rs b/kv-service/client-rs/src/rail_read.rs new file mode 100644 index 0000000..8f19332 --- /dev/null +++ b/kv-service/client-rs/src/rail_read.rs @@ -0,0 +1,925 @@ +//! Multi-rail descriptor reads for a single client Worker. + +use crate::pb; +use crate::rdma::{RdmaClient, RdmaClientConfig, RdmaReadOutcome}; +use std::collections::HashSet; +use std::fmt; +use std::path::Path; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +/// One local RDMA device and one listener on the node advertised by placement. +#[derive(Clone, Debug)] +pub struct RailRoute { + /// Stable name used in metrics and errors. + pub id: String, + /// Owning node's endpoint returned by `LookupObject`. + pub advertised_endpoint: String, + /// Local device and the listener actually dialed over this rail. + pub connection: RdmaClientConfig, + /// Initial availability; `RailReader::set_enabled` can change it later. + pub enabled: bool, + /// Relative capacity hint used by byte-weighted scheduling. + pub weight: u32, +} + +/// Local RDMA device placement reported by sysfs when available. +#[derive(Clone, Debug, Default, PartialEq, Eq)] +pub struct RailTopology { + /// NUMA node from sysfs, when the host reports one. + pub numa_node: Option, + /// PCI bus/device/function address of the local HCA. + pub pci_bdf: Option, +} + +fn read_topology_from(base: &Path, device: &str) -> RailTopology { + let path = base.join(device).join("device"); + let numa_node = std::fs::read_to_string(path.join("numa_node")) + .ok() + .and_then(|text| text.trim().parse::().ok()) + .filter(|node| *node >= 0); + let pci_bdf = std::fs::read_link(&path).ok().and_then(|target| { + target + .file_name() + .map(|name| name.to_string_lossy().into_owned()) + .filter(|name| { + let bytes = name.as_bytes(); + bytes.len() == 12 + && bytes[4] == b':' + && bytes[7] == b':' + && bytes[10] == b'.' + && bytes.iter().enumerate().all(|(index, byte)| { + matches!(index, 4 | 7 | 10) || byte.is_ascii_hexdigit() + }) + }) + }); + RailTopology { numa_node, pci_bdf } +} + +impl RailRoute { + /// Configure one local/remote path for an advertised storage endpoint. + pub fn new( + id: impl Into, + advertised_endpoint: impl Into, + connection: RdmaClientConfig, + ) -> Self { + Self { + id: id.into(), + advertised_endpoint: advertised_endpoint.into(), + connection, + enabled: true, + weight: 1, + } + } + + /// Keep a configured path visible while excluding it from read plans. + pub fn disabled(mut self) -> Self { + self.enabled = false; + self + } + + /// Set a positive relative scheduling weight for this path. + pub fn with_weight(mut self, weight: u32) -> Self { + self.weight = weight.max(1); + self + } +} + +/// Bounds all allocations and transfers made by one reader instance. +#[derive(Clone, Debug)] +pub struct RailLimits { + /// Maximum concurrent object reads accepted by this reader. + pub max_active_reads: usize, + /// Maximum final and per-rail staging allocation reserved at once. + pub max_staging_bytes: u64, + /// Maximum total MR lengths reserved across active reads. + pub max_registered_bytes: u64, + /// Maximum object payload bytes in flight at once. + pub max_inflight_bytes: u64, + /// Maximum payload bytes assigned to one rail for one request. + pub max_inflight_bytes_per_rail: u64, + /// TCP control deadline for connect, send, and reply operations. + pub io_timeout: Duration, + /// Delay before a failed rail is eligible for a later request. + pub rail_cooldown: Duration, +} + +impl Default for RailLimits { + fn default() -> Self { + Self { + max_active_reads: 8, + max_staging_bytes: 4 * 1024 * 1024 * 1024, + max_registered_bytes: 4 * 1024 * 1024 * 1024, + max_inflight_bytes: 4 * 1024 * 1024 * 1024, + max_inflight_bytes_per_rail: 2 * 1024 * 1024 * 1024, + io_timeout: Duration::from_secs(30), + rail_cooldown: Duration::from_secs(10), + } + } +} + +/// Typed failure returned without publishing a partial object. +#[derive(Debug)] +pub enum RailReadError { + InvalidPlacement(String), + ResourceExhausted(String), + BufferTooSmall { need: usize, have: usize }, + Transport(String), + StaleDescriptor, + Incomplete { expected: usize, actual: usize }, + Checksum { stripe: u32 }, + VersionChanged, + Cancelled, + WorkerPanic, +} + +impl fmt::Display for RailReadError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::InvalidPlacement(reason) => write!(f, "invalid placement: {reason}"), + Self::ResourceExhausted(reason) => write!(f, "rail resource limit: {reason}"), + Self::BufferTooSmall { need, have } => { + write!(f, "read buffer needs {need} bytes, has {have}") + } + Self::Transport(reason) => write!(f, "rail transfer failed: {reason}"), + Self::StaleDescriptor => write!(f, "object descriptor became stale"), + Self::Incomplete { expected, actual } => { + write!( + f, + "incomplete rail read: expected {expected}, received {actual}" + ) + } + Self::Checksum { stripe } => write!(f, "checksum mismatch on stripe {stripe}"), + Self::VersionChanged => write!(f, "object version changed during rail read"), + Self::Cancelled => write!(f, "rail read cancelled"), + Self::WorkerPanic => write!(f, "rail worker panicked"), + } + } +} + +impl std::error::Error for RailReadError {} + +pub(crate) fn same_descriptor_identity( + left: &pb::ObjectDescriptor, + right: &pb::ObjectDescriptor, +) -> bool { + left.key == right.key + && left.object_handle == right.object_handle + && left.object_generation == right.object_generation + && left.content_etag == right.content_etag + && left.layout_version == right.layout_version + && left.size == right.size + && left.is_striped == right.is_striped + && left.stripe_count == right.stripe_count + && left.chunk_size == right.chunk_size +} + +pub(crate) fn commit_if_unchanged( + initial: &crate::ObjectLookup, + current: &crate::ObjectLookup, + payload: &[u8], + destination: &mut [u8], +) -> Result { + if !same_descriptor_identity(&initial.descriptor, ¤t.descriptor) + || initial.placement != current.placement + { + return Err(RailReadError::VersionChanged); + } + let size = usize::try_from(initial.descriptor.size) + .map_err(|_| RailReadError::InvalidPlacement("object size exceeds address space".into()))?; + if payload.len() != size { + return Err(RailReadError::Incomplete { + expected: size, + actual: payload.len(), + }); + } + if destination.len() < size { + return Err(RailReadError::BufferTooSmall { + need: size, + have: destination.len(), + }); + } + destination[..size].copy_from_slice(payload); + Ok(size) +} + +/// Cooperative request cancellation shared with the caller. +#[derive(Clone, Default)] +pub struct RailCancel(Arc); + +impl RailCancel { + /// Request cancellation; already started rail workers still quiesce. + pub fn cancel(&self) { + self.0.store(true, Ordering::Release); + } + + /// Whether cancellation has been requested. + pub fn is_cancelled(&self) -> bool { + self.0.load(Ordering::Acquire) + } +} + +#[derive(Clone, Debug)] +struct StripeRead { + index: u32, + object_offset: usize, + length: usize, + packed_offset: usize, +} + +#[derive(Clone, Debug)] +struct RailTask { + route_index: usize, + stripes: Vec, + packed_len: usize, + dummy_len: usize, +} + +struct RailPlan { + size: usize, + chunk_size: usize, + checksums: Vec, + tasks: Vec, +} + +impl RailPlan { + fn build( + descriptor: &pb::ObjectDescriptor, + placement: &pb::PlacementDescriptor, + routes: &[RailRoute], + ) -> Result { + let size = usize::try_from(descriptor.size).map_err(|_| { + RailReadError::InvalidPlacement("object size exceeds address space".into()) + })?; + if descriptor.key.is_none() + || placement.key != descriptor.key + || descriptor.object_handle.is_empty() + || descriptor.object_generation == 0 + || descriptor.layout_version == 0 + { + return Err(RailReadError::InvalidPlacement( + "descriptor identity and placement key do not match".into(), + )); + } + if size == 0 { + return Err(RailReadError::InvalidPlacement("empty RDMA object".into())); + } + let (stripe_count, chunk_size) = if descriptor.is_striped { + let chunk_size = usize::try_from(descriptor.chunk_size).map_err(|_| { + RailReadError::InvalidPlacement("chunk size exceeds address space".into()) + })?; + if chunk_size == 0 || descriptor.stripe_count as usize != size.div_ceil(chunk_size) { + return Err(RailReadError::InvalidPlacement( + "descriptor stripe count and chunk size disagree".into(), + )); + } + (descriptor.stripe_count as usize, chunk_size) + } else { + (1, size) + }; + if stripe_count > u16::MAX as usize || placement.chunks.len() != stripe_count { + return Err(RailReadError::InvalidPlacement( + "placement stripe count exceeds the wire limit or is incomplete".into(), + )); + } + let mut ordered = vec![None; stripe_count]; + for chunk in &placement.chunks { + let index = chunk.stripe_index as usize; + if index >= stripe_count || ordered[index].is_some() { + return Err(RailReadError::InvalidPlacement(format!( + "stripe {} is duplicate or out of bounds", + chunk.stripe_index + ))); + } + let offset = index * chunk_size; + let length = (size - offset).min(chunk_size); + if chunk.offset != offset as u64 + || chunk.length != length as u64 + || chunk.rdma_endpoint.is_empty() + { + return Err(RailReadError::InvalidPlacement(format!( + "stripe {index} has a wrong offset, length, or endpoint" + ))); + } + ordered[index] = Some(chunk); + } + let mut tasks: Vec = routes + .iter() + .enumerate() + .filter(|(_, route)| route.enabled) + .map(|(route_index, _)| RailTask { + route_index, + stripes: Vec::new(), + packed_len: 0, + dummy_len: 0, + }) + .collect(); + let mut checksums = Vec::with_capacity(stripe_count); + for (index, chunk) in ordered.into_iter().enumerate() { + let chunk = chunk.ok_or_else(|| { + RailReadError::InvalidPlacement(format!("missing stripe {index}")) + })?; + let offset = index * chunk_size; + let length = (size - offset).min(chunk_size); + let selected = tasks + .iter() + .enumerate() + .filter(|(_, task)| { + routes[task.route_index].advertised_endpoint == chunk.rdma_endpoint + }) + .min_by(|(_, left), (_, right)| { + let left_weight = routes[left.route_index].weight as u128; + let right_weight = routes[right.route_index].weight as u128; + ((left.packed_len as u128) * right_weight) + .cmp(&((right.packed_len as u128) * left_weight)) + .then_with(|| right_weight.cmp(&left_weight)) + }) + .map(|(position, _)| position) + .ok_or_else(|| { + RailReadError::InvalidPlacement(format!( + "no enabled rail for {}", + chunk.rdma_endpoint + )) + })?; + let task = &mut tasks[selected]; + task.stripes.push(StripeRead { + index: index as u32, + object_offset: offset, + length, + packed_offset: task.packed_len, + }); + task.packed_len += length; + checksums.push(chunk.checksum.clone()); + } + tasks.retain(|task| !task.stripes.is_empty()); + let checksum_count = checksums + .iter() + .filter(|checksum| !checksum.is_empty()) + .count(); + if checksum_count != 0 && checksum_count != stripe_count { + return Err(RailReadError::InvalidPlacement( + "placement has only some stripe checksums".into(), + )); + } + if tasks.is_empty() { + return Err(RailReadError::InvalidPlacement("no rail selected".into())); + } + for task in &mut tasks { + task.dummy_len = if task.stripes.len() == stripe_count { + 0 + } else { + chunk_size.min(size) + }; + } + Ok(Self { + size, + chunk_size, + checksums, + tasks, + }) + } +} + +trait RailTransport: Sync { + fn fetch( + &self, + route: &RailRoute, + task: &RailTask, + descriptor: &pb::ObjectDescriptor, + timeout: Duration, + ) -> Result, RailReadError>; +} + +#[derive(Default)] +struct RailCounters { + enabled: AtomicBool, + cooldown_until: Mutex>, + reads_ok: AtomicU64, + reads_err: AtomicU64, + bytes: AtomicU64, + inflight_requests: AtomicU64, + inflight_bytes: AtomicU64, + peak_inflight_bytes: AtomicU64, + registered_bytes: AtomicU64, + peak_registered_bytes: AtomicU64, + duration_us: AtomicU64, +} + +impl RailCounters { + fn cooldown_remaining(&self) -> Duration { + self.cooldown_until + .lock() + .unwrap() + .map(|until| until.saturating_duration_since(Instant::now())) + .unwrap_or_default() + } + + fn cooldown_ms(&self) -> u64 { + self.cooldown_remaining().as_millis() as u64 + } + + fn is_available(&self) -> bool { + self.enabled.load(Ordering::Acquire) && self.cooldown_remaining().is_zero() + } +} + +struct RailActivityGuard<'a> { + counters: &'a RailCounters, + bytes: u64, + registered: u64, +} + +impl<'a> RailActivityGuard<'a> { + fn new(counters: &'a RailCounters, bytes: u64, registered: u64) -> Self { + counters.inflight_requests.fetch_add(1, Ordering::Relaxed); + let inflight = counters.inflight_bytes.fetch_add(bytes, Ordering::Relaxed) + bytes; + counters + .peak_inflight_bytes + .fetch_max(inflight, Ordering::Relaxed); + let pinned = counters + .registered_bytes + .fetch_add(registered, Ordering::Relaxed) + + registered; + counters + .peak_registered_bytes + .fetch_max(pinned, Ordering::Relaxed); + Self { + counters, + bytes, + registered, + } + } +} + +impl Drop for RailActivityGuard<'_> { + fn drop(&mut self) { + self.counters + .inflight_requests + .fetch_sub(1, Ordering::Relaxed); + self.counters + .inflight_bytes + .fetch_sub(self.bytes, Ordering::Relaxed); + self.counters + .registered_bytes + .fetch_sub(self.registered, Ordering::Relaxed); + } +} + +/// One rail's cumulative and current metrics. +#[derive(Clone, Debug)] +pub struct RailSnapshot { + /// Configured rail name. + pub id: String, + /// Dialed control listener. + pub listener: String, + /// Local Verbs device name. + pub device: String, + /// Cached NUMA and PCIe placement for the local device. + pub topology: RailTopology, + /// Whether this rail is enabled and outside its failure cooldown. + pub healthy: bool, + /// Milliseconds until a failed rail becomes selectable again. + pub cooldown_ms: u64, + /// Successful stripe-subset requests. + pub reads_ok: u64, + /// Failed stripe-subset requests. + pub reads_err: u64, + /// Verified payload bytes returned by this rail. + pub bytes: u64, + /// Requests currently executing on this rail. + pub inflight_requests: u64, + /// Payload bytes currently assigned to executing requests. + pub inflight_bytes: u64, + /// Maximum observed in-flight payload bytes. + pub peak_inflight_bytes: u64, + /// Reserved MR bytes while a task is active (not a hardware pin counter). + pub registered_bytes: u64, + /// Maximum observed reserved MR length. + pub peak_registered_bytes: u64, + /// Cumulative task duration in microseconds. + pub duration_us: u64, +} + +#[derive(Default)] +struct BudgetState { + active_reads: usize, + staging_bytes: u64, + registered_bytes: u64, + inflight_bytes: u64, +} + +/// Bounded multi-rail reader; one active request uses one QP per chosen rail. +pub struct RailReader { + routes: Vec, + topologies: Vec, + limits: RailLimits, + counters: Vec, + budget: Mutex, +} + +struct BudgetGuard<'a> { + reader: &'a RailReader, + staging: u64, + registered: u64, + inflight: u64, +} + +impl Drop for BudgetGuard<'_> { + fn drop(&mut self) { + let mut budget = self.reader.budget.lock().unwrap(); + budget.active_reads -= 1; + budget.staging_bytes -= self.staging; + budget.registered_bytes -= self.registered; + budget.inflight_bytes -= self.inflight; + } +} + +impl RailReader { + /// Validate routes and create an RDMA reader without opening connections. + pub fn new(routes: Vec, limits: RailLimits) -> Result { + if routes.is_empty() || limits.max_active_reads == 0 { + return Err(RailReadError::InvalidPlacement( + "at least one rail and one active read slot are required".into(), + )); + } + let mut ids = HashSet::new(); + let mut paths = HashSet::new(); + let mut local_ports = HashSet::new(); + for route in &routes { + if route.id.is_empty() + || route.advertised_endpoint.is_empty() + || route.connection.endpoint.is_empty() + || route.weight == 0 + || !ids.insert(route.id.clone()) + || !paths.insert(( + route.advertised_endpoint.clone(), + route.connection.endpoint.clone(), + )) + || !local_ports.insert(( + route.advertised_endpoint.clone(), + route.connection.device.clone(), + route.connection.port, + )) + { + return Err(RailReadError::InvalidPlacement( + "rail IDs, listeners, and local ports must identify distinct paths".into(), + )); + } + } + let topologies = routes + .iter() + .map(|route| { + read_topology_from(Path::new("/sys/class/infiniband"), &route.connection.device) + }) + .collect(); + let counters = routes + .iter() + .map(|route| RailCounters { + enabled: AtomicBool::new(route.enabled), + ..RailCounters::default() + }) + .collect(); + Ok(Self { + routes, + topologies, + limits, + counters, + budget: Mutex::new(BudgetState::default()), + }) + } + + /// Read per-rail counters without blocking in-flight transfers. + pub fn snapshots(&self) -> Vec { + self.routes + .iter() + .zip(&self.counters) + .enumerate() + .map(|(index, (route, counters))| RailSnapshot { + id: route.id.clone(), + listener: route.connection.endpoint.clone(), + device: route.connection.device.clone(), + topology: self.topologies[index].clone(), + healthy: counters.is_available(), + cooldown_ms: counters.cooldown_ms(), + reads_ok: counters.reads_ok.load(Ordering::Relaxed), + reads_err: counters.reads_err.load(Ordering::Relaxed), + bytes: counters.bytes.load(Ordering::Relaxed), + inflight_requests: counters.inflight_requests.load(Ordering::Relaxed), + inflight_bytes: counters.inflight_bytes.load(Ordering::Relaxed), + peak_inflight_bytes: counters.peak_inflight_bytes.load(Ordering::Relaxed), + registered_bytes: counters.registered_bytes.load(Ordering::Relaxed), + peak_registered_bytes: counters.peak_registered_bytes.load(Ordering::Relaxed), + duration_us: counters.duration_us.load(Ordering::Relaxed), + }) + .collect() + } + + /// Enable or disable one Rail without modifying object placement. + pub fn set_enabled(&self, id: &str, enabled: bool) -> bool { + let Some(index) = self.routes.iter().position(|route| route.id == id) else { + return false; + }; + let counters = &self.counters[index]; + counters.enabled.store(enabled, Ordering::Release); + if enabled { + *counters.cooldown_until.lock().unwrap() = None; + } + true + } + + fn reserve(&self, plan: &RailPlan) -> Result, RailReadError> { + let registered = plan + .tasks + .iter() + .try_fold(0u64, |sum, task| { + task.packed_len + .checked_add(task.dummy_len) + .and_then(|bytes| sum.checked_add(bytes as u64)) + }) + .ok_or_else(|| RailReadError::ResourceExhausted("registration size overflow".into()))?; + let staging = registered + .checked_add(plan.size as u64) + .ok_or_else(|| RailReadError::ResourceExhausted("staging size overflow".into()))?; + let inflight = plan.size as u64; + if plan + .tasks + .iter() + .any(|task| task.packed_len as u64 > self.limits.max_inflight_bytes_per_rail) + { + return Err(RailReadError::ResourceExhausted( + "per-rail in-flight bytes".into(), + )); + } + let mut budget = self.budget.lock().unwrap(); + if budget.active_reads >= self.limits.max_active_reads + || budget.staging_bytes.saturating_add(staging) > self.limits.max_staging_bytes + || budget.registered_bytes.saturating_add(registered) > self.limits.max_registered_bytes + || budget.inflight_bytes.saturating_add(inflight) > self.limits.max_inflight_bytes + { + return Err(RailReadError::ResourceExhausted( + "active reads, staging, registration, or in-flight bytes".into(), + )); + } + budget.active_reads += 1; + budget.staging_bytes += staging; + budget.registered_bytes += registered; + budget.inflight_bytes += inflight; + Ok(BudgetGuard { + reader: self, + staging, + registered, + inflight, + }) + } + + fn read_staged_with( + &self, + descriptor: &pb::ObjectDescriptor, + placement: &pb::PlacementDescriptor, + transport: &T, + cancel: Option<&RailCancel>, + ) -> Result, RailReadError> { + let mut available = self.routes.clone(); + for (route, counters) in available.iter_mut().zip(&self.counters) { + route.enabled = counters.is_available(); + } + let plan = RailPlan::build(descriptor, placement, &available)?; + let _budget = self.reserve(&plan)?; + if cancel.is_some_and(RailCancel::is_cancelled) { + return Err(RailReadError::Cancelled); + } + let results = std::thread::scope(|scope| { + let handles: Vec<_> = plan + .tasks + .iter() + .map(|task| { + let route = &self.routes[task.route_index]; + let counters = &self.counters[task.route_index]; + scope.spawn(move || { + let started = Instant::now(); + let _activity = RailActivityGuard::new( + counters, + task.packed_len as u64, + (task.packed_len + task.dummy_len) as u64, + ); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + transport.fetch(route, task, descriptor, self.limits.io_timeout) + })) + .unwrap_or(Err(RailReadError::WorkerPanic)); + counters + .duration_us + .fetch_add(started.elapsed().as_micros() as u64, Ordering::Relaxed); + if result.is_ok() { + counters.reads_ok.fetch_add(1, Ordering::Relaxed); + counters + .bytes + .fetch_add(task.packed_len as u64, Ordering::Relaxed); + } else { + counters.reads_err.fetch_add(1, Ordering::Relaxed); + if matches!( + &result, + Err(RailReadError::Transport(_) + | RailReadError::Incomplete { .. } + | RailReadError::WorkerPanic) + ) { + *counters.cooldown_until.lock().unwrap() = + Some(Instant::now() + self.limits.rail_cooldown); + } + } + result + }) + }) + .collect(); + handles + .into_iter() + .map(|handle| handle.join().unwrap_or(Err(RailReadError::WorkerPanic))) + .collect::>() + }); + if cancel.is_some_and(RailCancel::is_cancelled) { + return Err(RailReadError::Cancelled); + } + let mut staged = Vec::new(); + staged + .try_reserve_exact(plan.size) + .map_err(|error| RailReadError::ResourceExhausted(error.to_string()))?; + staged.resize(plan.size, 0); + for (task, result) in plan.tasks.iter().zip(results) { + let packed = result?; + if packed.len() != task.packed_len { + return Err(RailReadError::Incomplete { + expected: task.packed_len, + actual: packed.len(), + }); + } + for stripe in &task.stripes { + staged[stripe.object_offset..stripe.object_offset + stripe.length].copy_from_slice( + &packed[stripe.packed_offset..stripe.packed_offset + stripe.length], + ); + } + } + for (index, checksum) in plan.checksums.iter().enumerate() { + if checksum.is_empty() { + continue; + } + let start = index * plan.chunk_size; + let end = (start + plan.chunk_size).min(plan.size); + let actual = format!("{:016x}", twox_hash::xxh3::hash64(&staged[start..end])); + if !checksum.eq_ignore_ascii_case(&actual) { + return Err(RailReadError::Checksum { + stripe: index as u32, + }); + } + } + Ok(staged) + } + + fn read_into_with( + &self, + descriptor: &pb::ObjectDescriptor, + placement: &pb::PlacementDescriptor, + destination: &mut [u8], + transport: &T, + cancel: Option<&RailCancel>, + ) -> Result { + let size = usize::try_from(descriptor.size).map_err(|_| { + RailReadError::InvalidPlacement("object size exceeds address space".into()) + })?; + if destination.len() < size { + return Err(RailReadError::BufferTooSmall { + need: size, + have: destination.len(), + }); + } + let staged = self.read_staged_with(descriptor, placement, transport, cancel)?; + destination[..size].copy_from_slice(&staged); + Ok(size) + } + + /// Return a complete private object buffer after all rails and checksums pass. + /// Callers that require a stable version must perform a post-read lookup + /// before publishing it, as `KvClient::read_multi_rail_into` does. + pub fn read_staged( + &self, + descriptor: &pb::ObjectDescriptor, + placement: &pb::PlacementDescriptor, + cancel: Option<&RailCancel>, + ) -> Result, RailReadError> { + self.read_staged_with(descriptor, placement, &VerbsTransport, cancel) + } + + /// Copy a completed read into the caller's buffer. This low-level method + /// validates each server-side descriptor request but does not re-lookup the + /// object after transfer; use `KvClient::read_multi_rail_into` for that. + pub fn read_into( + &self, + descriptor: &pb::ObjectDescriptor, + placement: &pb::PlacementDescriptor, + destination: &mut [u8], + cancel: Option<&RailCancel>, + ) -> Result { + self.read_into_with(descriptor, placement, destination, &VerbsTransport, cancel) + } +} + +fn build_sge_segments( + task: &RailTask, + descriptor: &pb::ObjectDescriptor, + base: u64, + rkey: u32, +) -> Result, RailReadError> { + let registered_len = task + .packed_len + .checked_add(task.dummy_len) + .ok_or_else(|| RailReadError::InvalidPlacement("registered size overflow".into()))?; + let registered_end = base + .checked_add(registered_len as u64) + .ok_or_else(|| RailReadError::InvalidPlacement("registered address overflow".into()))?; + let mut segments = Vec::with_capacity(descriptor.stripe_count as usize); + let mut owned = task.stripes.iter().peekable(); + for index in 0..descriptor.stripe_count as usize { + let offset = index + .checked_mul(descriptor.chunk_size as usize) + .ok_or_else(|| RailReadError::InvalidPlacement("stripe offset overflow".into()))?; + let length = (descriptor.size as usize - offset).min(descriptor.chunk_size as usize); + let local_offset = if owned + .peek() + .is_some_and(|stripe| stripe.index as usize == index) + { + owned.next().expect("owned stripe").packed_offset + } else { + task.packed_len + }; + let address = base + .checked_add(local_offset as u64) + .ok_or_else(|| RailReadError::InvalidPlacement("SGE address overflow".into()))?; + if address + .checked_add(length as u64) + .is_none_or(|end| end > registered_end) + { + return Err(RailReadError::InvalidPlacement( + "SGE maps outside registered memory".into(), + )); + } + segments.push((address, rkey, length as u64)); + } + Ok(segments) +} + +struct VerbsTransport; + +impl RailTransport for VerbsTransport { + fn fetch( + &self, + route: &RailRoute, + task: &RailTask, + descriptor: &pb::ObjectDescriptor, + timeout: Duration, + ) -> Result, RailReadError> { + let mut client = RdmaClient::connect(route.connection.clone().with_io_timeout(timeout)) + .map_err(|error| RailReadError::Transport(error.to_string()))?; + let capacity = task + .packed_len + .checked_add(task.dummy_len) + .ok_or_else(|| RailReadError::ResourceExhausted("rail buffer overflow".into()))?; + let mut packed = Vec::new(); + packed + .try_reserve_exact(capacity) + .map_err(|error| RailReadError::ResourceExhausted(error.to_string()))?; + packed.resize(capacity, 0u8); + let registered = client + .register_buffer(&mut packed) + .map_err(|error| RailReadError::Transport(error.to_string()))?; + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + if descriptor.is_striped { + let view = registered.view(); + let segments = build_sge_segments(task, descriptor, view.addr(), view.rkey()) + .map_err(|error| anyhow::anyhow!(error.to_string()))?; + let indices: Vec = task.stripes.iter().map(|stripe| stripe.index).collect(); + client.get_descriptor_stripes_sge_detailed(descriptor, &indices, &segments) + } else { + client + .get_descriptor_into(descriptor, ®istered, 0) + .map(|outcome| outcome.map(|bytes| RdmaReadOutcome { bytes, chunks: 1 })) + } + })); + drop(client); + drop(registered); + let result = match result { + Ok(result) => result.map_err(|error| RailReadError::Transport(error.to_string()))?, + Err(panic) => std::panic::resume_unwind(panic), + }; + let outcome = result.ok_or(RailReadError::StaleDescriptor)?; + if outcome.bytes != task.packed_len || outcome.chunks != task.stripes.len() as u32 { + return Err(RailReadError::Incomplete { + expected: task.packed_len, + actual: outcome.bytes, + }); + } + packed.truncate(task.packed_len); + Ok(packed) + } +} + +#[cfg(test)] +mod tests; diff --git a/kv-service/client-rs/src/rail_read/tests.rs b/kv-service/client-rs/src/rail_read/tests.rs new file mode 100644 index 0000000..e6e2851 --- /dev/null +++ b/kv-service/client-rs/src/rail_read/tests.rs @@ -0,0 +1,686 @@ +use super::*; +use crate::pb; +use std::sync::Mutex; + +struct MockTransport { + object: Vec, + calls: Mutex>, + failing_rail: Option, +} + +impl MockTransport { + fn new(object: Vec) -> Self { + Self { + object, + calls: Mutex::new(Vec::new()), + failing_rail: None, + } + } +} + +impl RailTransport for MockTransport { + fn fetch( + &self, + route: &RailRoute, + task: &RailTask, + _descriptor: &pb::ObjectDescriptor, + _timeout: std::time::Duration, + ) -> Result, RailReadError> { + self.calls + .lock() + .unwrap() + .push(route.connection.endpoint.clone()); + let mut packed = vec![0u8; task.packed_len]; + for stripe in &task.stripes { + packed[stripe.packed_offset..stripe.packed_offset + stripe.length].copy_from_slice( + &self.object[stripe.object_offset..stripe.object_offset + stripe.length], + ); + } + if self.failing_rail.as_deref() == Some(&route.id) { + return Err(RailReadError::Transport("injected disconnect".into())); + } + Ok(packed) + } +} + +fn fixture(size: usize, chunk_size: usize) -> (pb::ObjectDescriptor, pb::PlacementDescriptor) { + let count = size.div_ceil(chunk_size); + let descriptor = pb::ObjectDescriptor { + key: Some(pb::ObjectKey { + namespace: "test".into(), + object_key: "obj".into(), + }), + object_handle: "handle".into(), + object_generation: 1, + content_etag: "etag".into(), + layout_version: 1, + size: size as u64, + is_striped: true, + stripe_count: count as u32, + chunk_size: chunk_size as u64, + }; + let placement = pb::PlacementDescriptor { + key: descriptor.key.clone(), + chunks: (0..count) + .map(|index| pb::PlacementChunk { + stripe_index: index as u32, + node_id: "node-a".into(), + grpc_endpoint: "10.0.0.1:50051".into(), + rdma_endpoint: "10.0.0.1:50053".into(), + device_id: 0, + storage_handle: format!("stripe-{index}"), + offset: (index * chunk_size) as u64, + length: (size - index * chunk_size).min(chunk_size) as u64, + checksum: String::new(), + }) + .collect(), + ..Default::default() + }; + (descriptor, placement) +} + +fn routes() -> Vec { + vec![ + RailRoute::new( + "rail0", + "10.0.0.1:50053", + crate::rdma::RdmaClientConfig::new("10.0.0.1:50053", "mock0"), + ), + RailRoute::new( + "rail1", + "10.0.0.1:50053", + crate::rdma::RdmaClientConfig::new("10.0.1.1:50054", "mock1"), + ), + ] +} + +#[test] +fn two_independent_listeners_restore_one_unmodified_placement() { + let bytes: Vec = (0..64).map(|index| index as u8).collect(); + let mock = MockTransport::new(bytes.clone()); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + assert_eq!( + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(), + 64 + ); + assert_eq!(destination, bytes); + let calls = mock.calls.lock().unwrap(); + assert_eq!(calls.len(), 2); + assert!(calls.contains(&"10.0.0.1:50053".to_string())); + assert!(calls.contains(&"10.0.1.1:50054".to_string())); + let snapshots = reader.snapshots(); + assert_eq!(snapshots[0].bytes, 32); + assert_eq!(snapshots[1].bytes, 32); + assert!(snapshots.iter().all(|rail| rail.inflight_requests == 0)); + assert!(snapshots.iter().all(|rail| rail.registered_bytes == 0)); + assert_eq!(snapshots[0].peak_inflight_bytes, 32); + assert!(snapshots.iter().all(|rail| rail.peak_registered_bytes > 0)); +} + +#[test] +fn partial_rail_failure_leaves_caller_buffer_unchanged() { + let mut mock = MockTransport::new(vec![0x42; 64]); + mock.failing_rail = Some("rail1".into()); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + assert_eq!(destination, vec![0xA5; 64]); + assert_eq!(mock.calls.lock().unwrap().len(), 2); +} + +#[test] +fn duplicate_stripe_is_rejected_before_transport() { + let mock = MockTransport::new(vec![0x42; 64]); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, mut placement) = fixture(64, 8); + placement.chunks[1].stripe_index = 0; + let mut destination = vec![0xA5; 64]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + assert!(mock.calls.lock().unwrap().is_empty()); + assert_eq!(destination, vec![0xA5; 64]); +} + +#[test] +fn registration_budget_rejects_before_transport() { + let mock = MockTransport::new(vec![0x42; 64]); + let limits = RailLimits { + max_registered_bytes: 16, + ..RailLimits::default() + }; + let reader = RailReader::new(routes(), limits).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + assert!(mock.calls.lock().unwrap().is_empty()); +} + +#[test] +fn descriptor_identity_rejects_generation_and_layout_changes() { + let (original, _) = fixture(64, 8); + let mut changed = original.clone(); + assert!(same_descriptor_identity(&original, &changed)); + changed.object_generation += 1; + assert!(!same_descriptor_identity(&original, &changed)); + changed = original.clone(); + changed.layout_version += 1; + assert!(!same_descriptor_identity(&original, &changed)); +} + +#[test] +fn two_rails_for_one_node_require_distinct_local_ports() { + let mut rails = routes(); + rails[1].connection.device = rails[0].connection.device.clone(); + assert!(RailReader::new(rails, RailLimits::default()).is_err()); +} + +#[test] +fn weighted_scheduler_assigns_more_stripes_to_faster_rail() { + let mut configured = routes(); + configured[1] = configured[1].clone().with_weight(3); + let (descriptor, placement) = fixture(80, 8); + let plan = RailPlan::build(&descriptor, &placement, &configured).unwrap(); + assert_eq!(plan.tasks.len(), 2); + assert!(plan.tasks[1].packed_len > plan.tasks[0].packed_len); + assert_eq!( + plan.tasks.iter().map(|task| task.packed_len).sum::(), + 80 + ); +} + +#[cfg(unix)] +#[test] +fn topology_reads_numa_and_pci_address_from_sysfs_shape() { + let temp = tempfile::tempdir().unwrap(); + let device = temp.path().join("mlx5_0"); + let pci = temp.path().join("0000:03:00.0"); + std::fs::create_dir_all(&device).unwrap(); + std::fs::create_dir_all(&pci).unwrap(); + std::fs::write(pci.join("numa_node"), "1\n").unwrap(); + std::os::unix::fs::symlink(&pci, device.join("device")).unwrap(); + let topology = read_topology_from(temp.path(), "mlx5_0"); + assert_eq!(topology.numa_node, Some(1)); + assert_eq!(topology.pci_bdf.as_deref(), Some("0000:03:00.0")); +} + +#[cfg(unix)] +#[test] +fn virtual_soft_roce_device_is_not_reported_as_pci_hardware() { + let temp = tempfile::tempdir().unwrap(); + let device = temp.path().join("rxe0"); + let virtual_device = temp.path().join("virtual-rxe0"); + std::fs::create_dir_all(&device).unwrap(); + std::fs::create_dir_all(&virtual_device).unwrap(); + std::os::unix::fs::symlink(&virtual_device, device.join("device")).unwrap(); + let topology = read_topology_from(temp.path(), "rxe0"); + assert_eq!(topology.pci_bdf, None); +} + +#[test] +fn one_rail_restores_the_same_object_layout() { + let bytes: Vec = (0..64).map(|index| index as u8).collect(); + let mock = MockTransport::new(bytes.clone()); + let reader = RailReader::new(vec![routes()[0].clone()], RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + assert_eq!( + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(), + 64 + ); + assert_eq!(destination, bytes); + assert_eq!(mock.calls.lock().unwrap().len(), 1); +} + +#[test] +fn disabled_second_rail_falls_back_to_first_listener() { + let bytes = vec![0x42; 64]; + let mock = MockTransport::new(bytes.clone()); + let mut configured = routes(); + configured[1] = configured[1].clone().disabled(); + let reader = RailReader::new(configured, RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + assert_eq!(destination, bytes); + assert_eq!(mock.calls.lock().unwrap().as_slice(), ["10.0.0.1:50053"]); +} + +#[test] +fn runtime_disable_and_enable_replans_without_changing_placement() { + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + assert!(reader.set_enabled("rail1", false)); + let mock = MockTransport::new(vec![0x42; 64]); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + assert_eq!(mock.calls.lock().unwrap().as_slice(), ["10.0.0.1:50053"]); + assert!(reader.set_enabled("rail1", true)); + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + assert_eq!(mock.calls.lock().unwrap().len(), 3); +} + +#[test] +fn failed_rail_cools_down_then_recovers_on_next_request() { + let limits = RailLimits { + rail_cooldown: Duration::from_millis(30), + ..RailLimits::default() + }; + let reader = RailReader::new(routes(), limits).unwrap(); + let mut failed = MockTransport::new(vec![0x42; 64]); + failed.failing_rail = Some("rail1".into()); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &failed, None) + .is_err()); + assert!(!reader.snapshots()[1].healthy); + + let recovered = MockTransport::new(vec![0x42; 64]); + reader + .read_into_with(&descriptor, &placement, &mut destination, &recovered, None) + .unwrap(); + assert_eq!(recovered.calls.lock().unwrap().len(), 1); + std::thread::sleep(Duration::from_millis(40)); + reader + .read_into_with(&descriptor, &placement, &mut destination, &recovered, None) + .unwrap(); + assert_eq!(recovered.calls.lock().unwrap().len(), 3); +} + +#[test] +fn stale_descriptor_does_not_mark_healthy_transport_as_down() { + struct StaleMock(MockTransport); + impl RailTransport for StaleMock { + fn fetch( + &self, + route: &RailRoute, + task: &RailTask, + descriptor: &pb::ObjectDescriptor, + timeout: Duration, + ) -> Result, RailReadError> { + if route.id == "rail1" { + Err(RailReadError::StaleDescriptor) + } else { + self.0.fetch(route, task, descriptor, timeout) + } + } + } + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + let transport = StaleMock(MockTransport::new(vec![0x42; 64])); + assert!(matches!( + reader.read_into_with(&descriptor, &placement, &mut destination, &transport, None), + Err(RailReadError::StaleDescriptor) + )); + assert!(reader.snapshots().iter().all(|rail| rail.healthy)); +} + +#[test] +fn checksum_failure_keeps_destination_unchanged() { + let mock = MockTransport::new(vec![0x42; 64]); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, mut placement) = fixture(64, 8); + for chunk in &mut placement.chunks { + chunk.checksum = "0000000000000000".into(); + } + let mut destination = vec![0xA5; 64]; + assert!(matches!( + reader.read_into_with(&descriptor, &placement, &mut destination, &mock, None), + Err(RailReadError::Checksum { stripe: 0 }) + )); + assert_eq!(destination, vec![0xA5; 64]); +} + +#[test] +fn matching_stripe_checksums_allow_complete_read() { + let payload: Vec = (0..64).map(|index| index as u8).collect(); + let mock = MockTransport::new(payload.clone()); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, mut placement) = fixture(64, 8); + for chunk in &mut placement.chunks { + let start = chunk.offset as usize; + let end = start + chunk.length as usize; + chunk.checksum = format!("{:016x}", twox_hash::xxh3::hash64(&payload[start..end])); + } + let mut destination = vec![0xA5; 64]; + assert_eq!( + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(), + 64 + ); + assert_eq!(destination, payload); +} + +#[test] +fn missing_or_out_of_bounds_stripe_is_rejected_before_dispatch() { + let mock = MockTransport::new(vec![0x42; 64]); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, mut placement) = fixture(64, 8); + placement.chunks.pop(); + let mut destination = vec![0xA5; 64]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + let (_, mut placement) = fixture(64, 8); + placement.chunks[1].offset = 99; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + assert!(mock.calls.lock().unwrap().is_empty()); +} + +#[test] +fn placement_for_another_object_is_rejected_before_dispatch() { + let mock = MockTransport::new(vec![0x42; 64]); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, mut placement) = fixture(64, 8); + placement.key.as_mut().unwrap().object_key = "different".into(); + let mut destination = vec![0xA5; 64]; + assert!(matches!( + reader.read_into_with(&descriptor, &placement, &mut destination, &mock, None), + Err(RailReadError::InvalidPlacement(_)) + )); + assert!(mock.calls.lock().unwrap().is_empty()); +} + +#[test] +fn partially_populated_checksums_are_rejected_before_dispatch() { + let mock = MockTransport::new(vec![0x42; 64]); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, mut placement) = fixture(64, 8); + placement.chunks[0].checksum = "0000000000000000".into(); + let mut destination = vec![0xA5; 64]; + assert!(matches!( + reader.read_into_with(&descriptor, &placement, &mut destination, &mock, None), + Err(RailReadError::InvalidPlacement(_)) + )); + assert!(mock.calls.lock().unwrap().is_empty()); +} + +#[test] +fn cancellation_waits_for_started_mock_transfer_and_preserves_buffer() { + use std::sync::mpsc; + struct SlowMock { + base: MockTransport, + started: Mutex>>, + } + impl RailTransport for SlowMock { + fn fetch( + &self, + route: &RailRoute, + task: &RailTask, + descriptor: &pb::ObjectDescriptor, + timeout: Duration, + ) -> Result, RailReadError> { + if let Some(sender) = self.started.lock().unwrap().take() { + let _ = sender.send(()); + } + std::thread::sleep(Duration::from_millis(60)); + self.base.fetch(route, task, descriptor, timeout) + } + } + let (started_tx, started_rx) = mpsc::channel(); + let mock = SlowMock { + base: MockTransport::new(vec![0x42; 64]), + started: Mutex::new(Some(started_tx)), + }; + let token = RailCancel::default(); + let cancelling = token.clone(); + let canceller = std::thread::spawn(move || { + started_rx.recv().unwrap(); + cancelling.cancel(); + }); + let reader = RailReader::new(routes(), RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + let started = Instant::now(); + assert!(matches!( + reader.read_into_with( + &descriptor, + &placement, + &mut destination, + &mock, + Some(&token) + ), + Err(RailReadError::Cancelled) + )); + canceller.join().unwrap(); + assert!(started.elapsed() >= Duration::from_millis(60)); + assert_eq!(destination, vec![0xA5; 64]); + destination.fill(0x33); + std::thread::sleep(Duration::from_millis(20)); + assert_eq!(destination, vec![0x33; 64]); +} + +#[test] +fn sge_map_keeps_every_stripe_inside_its_registered_region() { + let (descriptor, placement) = fixture(64, 8); + let plan = RailPlan::build(&descriptor, &placement, &routes()).unwrap(); + let task = &plan.tasks[0]; + let base = 0x1000u64; + let segments = build_sge_segments(task, &descriptor, base, 7).unwrap(); + assert_eq!(segments.len(), 8); + assert_eq!(segments.iter().map(|(_, _, len)| len).sum::(), 64); + for (address, rkey, length) in &segments { + assert_eq!(*rkey, 7); + assert!(*address >= base); + assert!(*address + *length <= base + (task.packed_len + task.dummy_len) as u64); + } + assert_eq!(segments[0].0, base); + assert_eq!(segments[1].0, base + task.packed_len as u64); + assert_eq!(segments[2].0, base + 8); +} + +#[test] +fn changed_generation_cannot_publish_completed_rail_bytes() { + let (descriptor, placement) = fixture(64, 8); + let initial = crate::ObjectLookup { + descriptor: descriptor.clone(), + placement: Some(placement.clone()), + }; + let mut current = initial.clone(); + current.descriptor.object_generation += 1; + let mut destination = vec![0xA5; 64]; + assert!(matches!( + commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination), + Err(RailReadError::VersionChanged) + )); + assert_eq!(destination, vec![0xA5; 64]); + current = initial.clone(); + current.placement.as_mut().unwrap().layout_hash = "moved".into(); + assert!(matches!( + commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination), + Err(RailReadError::VersionChanged) + )); + assert_eq!(destination, vec![0xA5; 64]); + current = initial.clone(); + assert_eq!( + commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination).unwrap(), + 64 + ); + assert_eq!(destination, vec![0x42; 64]); +} + +#[test] +fn panicked_worker_releases_inflight_metrics_and_preserves_buffer() { + struct PanicTransport; + impl RailTransport for PanicTransport { + fn fetch( + &self, + _route: &RailRoute, + _task: &RailTask, + _descriptor: &pb::ObjectDescriptor, + _timeout: Duration, + ) -> Result, RailReadError> { + panic!("injected worker panic") + } + } + let reader = RailReader::new(vec![routes()[0].clone()], RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let mut destination = vec![0xA5; 64]; + assert!(matches!( + reader.read_into_with( + &descriptor, + &placement, + &mut destination, + &PanicTransport, + None + ), + Err(RailReadError::WorkerPanic) + )); + assert_eq!(reader.snapshots()[0].inflight_bytes, 0); + assert_eq!(destination, vec![0xA5; 64]); +} + +#[test] +fn active_read_limit_rejects_concurrent_second_request() { + use std::sync::mpsc; + struct HeldTransport { + object: MockTransport, + started: Mutex>>, + release: Mutex>, + } + impl RailTransport for HeldTransport { + fn fetch( + &self, + route: &RailRoute, + task: &RailTask, + descriptor: &pb::ObjectDescriptor, + timeout: Duration, + ) -> Result, RailReadError> { + if let Some(sender) = self.started.lock().unwrap().take() { + let _ = sender.send(()); + } + self.release.lock().unwrap().recv().unwrap(); + self.object.fetch(route, task, descriptor, timeout) + } + } + let limits = RailLimits { + max_active_reads: 1, + ..RailLimits::default() + }; + let reader = Arc::new(RailReader::new(routes(), limits).unwrap()); + let (descriptor, placement) = fixture(64, 8); + let (started_tx, started_rx) = mpsc::channel(); + let (release_tx, release_rx) = mpsc::channel(); + let held = HeldTransport { + object: MockTransport::new(vec![0x42; 64]), + started: Mutex::new(Some(started_tx)), + release: Mutex::new(release_rx), + }; + let first_reader = Arc::clone(&reader); + let first_desc = descriptor.clone(); + let first_placement = placement.clone(); + let first = std::thread::spawn(move || { + let mut destination = vec![0xA5; 64]; + first_reader + .read_into_with(&first_desc, &first_placement, &mut destination, &held, None) + .map(|_| destination) + }); + started_rx.recv().unwrap(); + let second_mock = MockTransport::new(vec![0x42; 64]); + let mut second_destination = vec![0xA5; 64]; + assert!(matches!( + reader.read_into_with( + &descriptor, + &placement, + &mut second_destination, + &second_mock, + None + ), + Err(RailReadError::ResourceExhausted(_)) + )); + assert!(second_mock.calls.lock().unwrap().is_empty()); + release_tx.send(()).unwrap(); + release_tx.send(()).unwrap(); + assert_eq!(first.join().unwrap().unwrap(), vec![0x42; 64]); + assert_eq!(second_destination, vec![0xA5; 64]); +} + +#[test] +#[ignore = "software-only Mock measurement; this is not RDMA bandwidth"] +fn software_only_mock_benchmark() { + fn process_usage() -> (u64, u64, u64) { + let mut usage = unsafe { std::mem::zeroed::() }; + assert_eq!(unsafe { libc::getrusage(libc::RUSAGE_SELF, &mut usage) }, 0); + let micros = |time: libc::timeval| time.tv_sec as u64 * 1_000_000 + time.tv_usec as u64; + ( + micros(usage.ru_utime), + micros(usage.ru_stime), + usage.ru_maxrss as u64, + ) + } + let env_number = |name: &str, default: usize| { + std::env::var(name) + .ok() + .and_then(|value| value.parse().ok()) + .unwrap_or(default) + }; + let mib = env_number("CS_RAIL_MOCK_MIB", 64); + let iterations = env_number("CS_RAIL_MOCK_ITERS", 5); + let rail_count = env_number("CS_RAIL_MOCK_RAILS", 2); + assert!(mib > 0 && iterations > 0 && (1..=2).contains(&rail_count)); + let size = mib * 1024 * 1024; + let object = vec![0x42; size]; + let mock = MockTransport::new(object); + let reader = RailReader::new(routes()[..rail_count].to_vec(), RailLimits::default()).unwrap(); + let (descriptor, placement) = fixture(size, 4 * 1024 * 1024); + let mut destination = vec![0xA5; size]; + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + let warmup_bytes: Vec<_> = reader + .snapshots() + .iter() + .map(|snapshot| snapshot.bytes) + .collect(); + let mut durations = Vec::with_capacity(iterations); + let mut cpu_user_us = 0u64; + let mut cpu_system_us = 0u64; + for _ in 0..iterations { + destination.fill(0xA5); + let before_cpu = process_usage(); + let start = Instant::now(); + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + durations.push(start.elapsed().as_micros() as u64); + let after_cpu = process_usage(); + cpu_user_us += after_cpu.0 - before_cpu.0; + cpu_system_us += after_cpu.1 - before_cpu.1; + assert!(destination.iter().all(|byte| *byte == 0x42)); + } + let average = durations.iter().sum::() / iterations as u64; + println!("mock_samples_us={durations:?}"); + let gib_per_s = size as f64 * 1_000_000.0 / average as f64 / 1024f64.powi(3); + let bytes: Vec<_> = reader + .snapshots() + .iter() + .zip(warmup_bytes) + .map(|(snapshot, warmup)| snapshot.bytes - warmup) + .collect(); + println!( + "mock_only,rails={rail_count},size_bytes={size},iters={iterations},avg_us={average},gib_per_s={gib_per_s:.3},cpu_user_us={cpu_user_us},cpu_system_us={cpu_system_us},peak_rss_kb={},rail_bytes={bytes:?}", + process_usage().2 + ); +} diff --git a/kv-service/client-rs/src/rdma.rs b/kv-service/client-rs/src/rdma.rs index a7532d0..f6c6f9a 100644 --- a/kv-service/client-rs/src/rdma.rs +++ b/kv-service/client-rs/src/rdma.rs @@ -16,7 +16,7 @@ use rdma_sys::*; use std::ffi::{c_void, CStr}; use std::io::{Read, Write}; use std::marker::PhantomData; -use std::net::TcpStream; +use std::net::{TcpStream, ToSocketAddrs}; use std::ptr::{self, NonNull}; use std::sync::Arc; use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; @@ -59,6 +59,8 @@ pub struct RdmaClientConfig { pub port: u8, /// GID index used to construct the RoCE address handle. pub gid_index: u8, + /// Deadline for TCP connect, control replies, and control writes. + pub io_timeout: Duration, } impl RdmaClientConfig { @@ -69,6 +71,7 @@ impl RdmaClientConfig { device: device.into(), port: 1, gid_index: 3, + io_timeout: Duration::from_secs(30), } } @@ -83,11 +86,57 @@ impl RdmaClientConfig { self.gid_index = gid_index; self } + + /// Bound TCP connection and control-message operations. + pub fn with_io_timeout(mut self, timeout: Duration) -> Self { + self.io_timeout = timeout; + self + } +} + +fn open_control_stream(endpoint: &str, timeout: Duration) -> Result { + if timeout.is_zero() { + return Err(anyhow!("RDMA control timeout must be positive")); + } + let mut last_error = None; + for address in endpoint + .to_socket_addrs() + .with_context(|| format!("resolve RDMA control endpoint {endpoint}"))? + { + match TcpStream::connect_timeout(&address, timeout) { + Ok(stream) => { + stream.set_nodelay(true).context("set RDMA TCP_NODELAY")?; + stream + .set_read_timeout(Some(timeout)) + .context("set RDMA control read timeout")?; + stream + .set_write_timeout(Some(timeout)) + .context("set RDMA control write timeout")?; + return Ok(stream); + } + Err(error) => last_error = Some(error), + } + } + Err(anyhow!( + "connect RDMA control endpoint {endpoint}: {}", + last_error + .map(|error| error.to_string()) + .unwrap_or_else(|| "no resolved addresses".to_string()) + )) } /// Result of an RDMA read. `None` means the object did not exist. pub type RdmaReadResult = Option; +/// Complete GET acknowledgement from the server, including its chunk count. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RdmaReadOutcome { + /// Payload bytes acknowledged by the server. + pub bytes: usize, + /// Number of stripes/chunks acknowledged by the server. + pub chunks: u32, +} + /// A connected ContextStore RDMA client. /// /// Every method takes `&mut self` because a connection serializes its TCP @@ -296,13 +345,7 @@ impl RdmaClient { } let result = (|| -> Result { - let mut stream = TcpStream::connect(&config.endpoint) - .with_context(|| format!("connect RDMA control endpoint {}", config.endpoint))?; - // 控制面小包必须即时发出: Nagle + delayed-ACK 在多连接并发时会给每个 - // 请求注入 ~40-200ms 延迟 (数据面 RDMA WRITE 不经 TCP, 不受影响). - stream - .set_nodelay(true) - .context("set TCP_NODELAY on RDMA control stream")?; + let mut stream = open_control_stream(&config.endpoint, config.io_timeout)?; let local = QpInfo { qpn: unsafe { (*qp.as_ptr()).qp_num }, psn: random_psn(), @@ -547,6 +590,17 @@ impl RdmaClient { stripes: &[u32], segments: &[(u64, u32, u64)], ) -> Result { + self.get_descriptor_stripes_sge_detailed(descriptor, stripes, segments) + .map(|outcome| outcome.map(|outcome| outcome.bytes)) + } + + /// SGE stripe GET with server-reported bytes and chunk count. + pub fn get_descriptor_stripes_sge_detailed( + &mut self, + descriptor: &pb::ObjectDescriptor, + stripes: &[u32], + segments: &[(u64, u32, u64)], + ) -> Result> { if segments.is_empty() { return Err(anyhow!("SGE GET requires at least one destination segment")); } @@ -579,7 +633,7 @@ impl RdmaClient { } self.stream.write_all(&request)?; self.stream.flush()?; - read_get_response(&mut self.stream) + read_get_response_detailed(&mut self.stream) } /// Write `buffer[offset..offset + size]` through the RDMA PUT data path. @@ -1137,6 +1191,10 @@ fn read_string(stream: &mut TcpStream, field: &str) -> Result { } fn read_get_response(stream: &mut TcpStream) -> Result { + read_get_response_detailed(stream).map(|outcome| outcome.map(|outcome| outcome.bytes)) +} + +fn read_get_response_detailed(stream: &mut TcpStream) -> Result> { let mut tag = [0u8; 1]; stream.read_exact(&mut tag)?; if tag[0] != MSG_GET_RESP { @@ -1149,7 +1207,8 @@ fn read_get_response(stream: &mut TcpStream) -> Result { } let bytes = u64::from_le_bytes(body[1..9].try_into().expect("fixed get response length")); let bytes = usize::try_from(bytes).map_err(|_| anyhow!("RDMA read size exceeds usize"))?; - Ok(Some(bytes)) + let chunks = u32::from_le_bytes(body[9..13].try_into().expect("fixed get response length")); + Ok(Some(RdmaReadOutcome { bytes, chunks })) } struct PutReady { @@ -1288,7 +1347,7 @@ mod tests { fn request_rejects_oversized_wire_string() { let key = "x".repeat(u16::MAX as usize + 1); assert!(build_get_request(&key, 1, 2, 3).is_err()); - assert!(build_put_request(MSG_PUT_REQ, &key, 3).is_err()); + assert!(build_put_request(MSG_PUT_REQ, &key, 3, 0).is_err()); } #[test] @@ -1297,4 +1356,36 @@ mod tests { assert_eq!(config.port, 1); assert_eq!(config.gid_index, 3); } + + #[test] + fn descriptor_reply_preserves_server_chunk_count() { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let endpoint = listener.local_addr().unwrap(); + let writer = std::thread::spawn(move || { + let (mut stream, _) = listener.accept().unwrap(); + let mut frame = vec![MSG_GET_RESP, 1]; + frame.extend_from_slice(&8u64.to_le_bytes()); + frame.extend_from_slice(&2u32.to_le_bytes()); + stream.write_all(&frame).unwrap(); + }); + let mut stream = TcpStream::connect(endpoint).unwrap(); + let outcome = read_get_response_detailed(&mut stream).unwrap().unwrap(); + assert_eq!(outcome.bytes, 8); + assert_eq!(outcome.chunks, 2); + writer.join().unwrap(); + } + + #[test] + fn control_socket_timeout_bounds_missing_get_reply() { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + let endpoint = listener.local_addr().unwrap(); + let server = std::thread::spawn(move || { + let (_stream, _) = listener.accept().unwrap(); + std::thread::sleep(Duration::from_millis(100)); + }); + let mut stream = + open_control_stream(&endpoint.to_string(), Duration::from_millis(20)).unwrap(); + assert!(read_get_response_detailed(&mut stream).is_err()); + server.join().unwrap(); + } } diff --git a/kv-service/client-rs/tests/local_cluster_e2e.rs b/kv-service/client-rs/tests/local_cluster_e2e.rs index a361147..4109c0d 100644 --- a/kv-service/client-rs/tests/local_cluster_e2e.rs +++ b/kv-service/client-rs/tests/local_cluster_e2e.rs @@ -219,12 +219,13 @@ rdma_endpoint = "" .unwrap(); let deadline = Instant::now() + Duration::from_secs(20); while Instant::now() < deadline { - assert!( - child.try_wait().unwrap().is_none(), - "{} exited: {}", - node, - fs::read_to_string(&log_path).unwrap_or_default() - ); + if let Some(status) = child.try_wait().unwrap() { + panic!( + "{} exited ({status}): {}", + node, + fs::read_to_string(&log_path).unwrap_or_default() + ); + } if let Ok(mut client) = KvClient::connect(format!("http://{}", self.endpoint(node))).await { @@ -234,6 +235,8 @@ rdma_endpoint = "" } tokio::time::sleep(Duration::from_millis(100)).await; } + let _ = child.kill(); + let _ = child.wait(); panic!( "{} did not become healthy: {}", node, diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs new file mode 100644 index 0000000..b456c79 --- /dev/null +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -0,0 +1,202 @@ +//! Hardware-gated checks for one Worker reading one object over two RDMA rails. +//! +//! Configure a striped KVService with two independent listeners, then set +//! `CS_RAIL_COORDINATOR`, `CS_RAIL_LISTENER0/1`, and `CS_RAIL_DEVICE0/1`. + +#![cfg(feature = "rdma")] + +use contextstore_client_rs::rail_read::{RailLimits, RailReader, RailRoute}; +use contextstore_client_rs::rdma::RdmaClientConfig; +use contextstore_client_rs::KvClient; +use prost::bytes::Bytes; +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +fn setting(name: &str, default: &str) -> String { + std::env::var(name).unwrap_or_else(|_| default.to_string()) +} + +fn route(advertised: &str, index: usize, listener: &str) -> RailRoute { + let device = setting(&format!("CS_RAIL_DEVICE{index}"), &format!("rxe{index}")); + let gid = setting(&format!("CS_RAIL_GID{index}"), "3") + .parse::() + .expect("GID index"); + RailRoute::new( + format!("rail{index}"), + advertised, + RdmaClientConfig::new(listener, device).with_gid_index(gid), + ) +} + +async fn seeded_object() -> (KvClient, String, Vec, String) { + let coordinator = setting("CS_RAIL_COORDINATOR", "http://127.0.0.1:50051"); + let mut client = KvClient::connect(coordinator).await.expect("connect gRPC"); + let nanos = SystemTime::now() + .duration_since(UNIX_EPOCH) + .expect("wall clock") + .as_nanos(); + let key = format!("rail-e2e-{nanos}"); + let mib = setting("CS_RAIL_OBJECT_MIB", "64") + .parse::() + .expect("object size in MiB"); + let payload: Vec = (0..mib * 1024 * 1024) + .map(|offset| (offset % 251) as u8) + .collect(); + let bytes = Bytes::from(payload.clone()); + let segment_size = 4 * 1024 * 1024; + let segments = (0..bytes.len()) + .step_by(segment_size) + .map(|offset| bytes.slice(offset..(offset + segment_size).min(bytes.len()))) + .collect(); + client + .put_stream_chunks("rail-e2e", &key, segments) + .await + .expect("seed object"); + let lookup = client + .lookup_object("rail-e2e", &key) + .await + .expect("lookup") + .expect("seeded object exists"); + assert!( + lookup.descriptor.is_striped && lookup.descriptor.stripe_count >= 2, + "configure striped storage and seed an object above its threshold" + ); + let advertised = lookup + .placement + .as_ref() + .expect("placement") + .chunks + .first() + .expect("stripe") + .rdma_endpoint + .clone(); + assert!( + lookup + .placement + .as_ref() + .expect("placement") + .chunks + .iter() + .all(|chunk| chunk.rdma_endpoint == advertised), + "this test requires one storage node with two RDMA listeners" + ); + (client, key, payload, advertised) +} + +#[tokio::test] +#[ignore = "requires two reachable RDMA listeners on the same storage node"] +async fn same_object_matches_single_and_dual_rail() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener0 = setting("CS_RAIL_LISTENER0", &advertised); + let listener1 = setting("CS_RAIL_LISTENER1", "127.0.0.1:50054"); + let limits = RailLimits { + io_timeout: Duration::from_secs(10), + ..RailLimits::default() + }; + let dual = Arc::new( + RailReader::new( + vec![ + route(&advertised, 0, &listener0), + route(&advertised, 1, &listener1), + ], + limits.clone(), + ) + .expect("dual reader"), + ); + let mut dual_bytes = vec![0xA5; payload.len()]; + assert_eq!( + client + .read_multi_rail_into(Arc::clone(&dual), "rail-e2e", &key, &mut dual_bytes, None) + .await + .expect("dual read"), + Some(payload.len()) + ); + assert_eq!(dual_bytes, payload); + assert!(dual.snapshots().iter().all(|snapshot| snapshot.bytes > 0)); + + let single = Arc::new( + RailReader::new(vec![route(&advertised, 0, &listener0)], limits).expect("single reader"), + ); + let mut single_bytes = vec![0x5A; payload.len()]; + assert_eq!( + client + .read_multi_rail_into(single, "rail-e2e", &key, &mut single_bytes, None) + .await + .expect("single read"), + Some(payload.len()) + ); + assert_eq!(single_bytes, dual_bytes); +} + +#[tokio::test] +#[ignore = "requires one reachable RDMA listener and one injected dead listener"] +async fn failed_second_rail_does_not_publish_partial_bytes() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener0 = setting("CS_RAIL_LISTENER0", &advertised); + let dead = setting("CS_RAIL_DEAD_LISTENER", "127.0.0.1:59999"); + let limits = RailLimits { + io_timeout: Duration::from_secs(3), + ..RailLimits::default() + }; + let reader = Arc::new( + RailReader::new( + vec![ + route(&advertised, 0, &listener0), + route(&advertised, 1, &dead), + ], + limits, + ) + .expect("reader"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert!(client + .read_multi_rail_into(reader, "rail-e2e", &key, &mut destination, None) + .await + .is_err()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} + +#[tokio::test] +#[ignore = "requires two reachable RDMA listeners and a striped KVService"] +async fn old_generation_is_rejected_without_publishing_bytes() { + let (mut client, key, mut payload, advertised) = seeded_object().await; + let old = client + .lookup_object("rail-e2e", &key) + .await + .expect("lookup") + .expect("seeded object"); + assert!(client.delete("rail-e2e", &key).await.expect("delete")); + payload[0] ^= 0xFF; + let bytes = Bytes::from(payload.clone()); + let segment_size = 4 * 1024 * 1024; + let segments = (0..bytes.len()) + .step_by(segment_size) + .map(|offset| bytes.slice(offset..(offset + segment_size).min(bytes.len()))) + .collect(); + client + .put_stream_chunks("rail-e2e", &key, segments) + .await + .expect("rewrite"); + let reader = RailReader::new( + vec![ + route(&advertised, 0, &setting("CS_RAIL_LISTENER0", &advertised)), + route( + &advertised, + 1, + &setting("CS_RAIL_LISTENER1", "127.0.0.1:50054"), + ), + ], + RailLimits::default(), + ) + .expect("reader"); + let mut destination = vec![0xA5; payload.len()]; + assert!(reader + .read_into( + &old.descriptor, + old.placement.as_ref().expect("old placement"), + &mut destination, + None, + ) + .is_err()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} From 9328cd42256ab6325fd3489f55ab98b6f556819c Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 00:49:01 +0800 Subject: [PATCH 02/20] fix O_DIRECT flag on Linux arm64 Use one architecture-aware flag for Tier A and Tier B. The previous x86 literal maps to O_DIRECTORY on arm64 and made striped reads fail with ENOTDIR; regression tests compare the chosen value with native libc on each Linux build. --- kv-service/server/src/io_executor/mod.rs | 9 ++++++++ kv-service/server/src/io_executor/tier_a.rs | 14 ++++++------- kv-service/server/src/io_executor/tier_b.rs | 23 ++++++++++++--------- kv-service/server/src/rdma/wire.rs | 1 - 4 files changed, 29 insertions(+), 18 deletions(-) diff --git a/kv-service/server/src/io_executor/mod.rs b/kv-service/server/src/io_executor/mod.rs index 36a07b2..59be212 100644 --- a/kv-service/server/src/io_executor/mod.rs +++ b/kv-service/server/src/io_executor/mod.rs @@ -10,6 +10,15 @@ mod aligned_buffer; mod tier_a; +// Linux arm64 assigns O_DIRECT a different bit from x86. Both executors use +// this one constant so a future architecture port has one value to verify. +#[cfg(all(target_os = "linux", target_arch = "aarch64"))] +const O_DIRECT_FLAG: i32 = 0o200000; +#[cfg(all(target_os = "linux", not(target_arch = "aarch64")))] +const O_DIRECT_FLAG: i32 = 0o40000; +#[cfg(not(target_os = "linux"))] +const O_DIRECT_FLAG: i32 = 0; + pub use aligned_buffer::{AlignedBuffer, PooledAlignedBuffer, READ_BUFFER_POOL}; #[cfg(all(feature = "io-uring", target_os = "linux"))] diff --git a/kv-service/server/src/io_executor/tier_a.rs b/kv-service/server/src/io_executor/tier_a.rs index 06abfd7..7c8816c 100644 --- a/kv-service/server/src/io_executor/tier_a.rs +++ b/kv-service/server/src/io_executor/tier_a.rs @@ -6,6 +6,7 @@ use super::{ log_io_batch, log_io_error, AlignedBuffer, IOExecutor, IORequest, IoBatchStats, IoLogContext, + O_DIRECT_FLAG, }; use crate::error::{KVError, Result}; use crossbeam_channel as channel; @@ -13,13 +14,6 @@ use prost::bytes::Bytes; use std::path::Path; use std::thread; -/// Linux O_DIRECT flag value (x86_64 / aarch64 = 0o40000); fully equivalent to libc::O_DIRECT. -/// Inlined as a constant to avoid depending on libc (CLAUDE.md: don't modify Cargo.toml deps). -#[cfg(target_os = "linux")] -const O_DIRECT_FLAG: i32 = 0o40000; -#[cfg(not(target_os = "linux"))] -const O_DIRECT_FLAG: i32 = 0; - /// Alignment required by O_DIRECT (Linux standard page size = filesystem block size = NVMe physical sector). pub const DIRECT_IO_ALIGN: usize = 4096; @@ -835,6 +829,12 @@ mod tests { use super::*; use tempfile::TempDir; + #[cfg(target_os = "linux")] + #[test] + fn direct_io_flag_matches_native_platform() { + assert_eq!(O_DIRECT_FLAG, libc::O_DIRECT); + } + #[test] fn write_read_roundtrip() { let tmp = TempDir::new().unwrap(); diff --git a/kv-service/server/src/io_executor/tier_b.rs b/kv-service/server/src/io_executor/tier_b.rs index 59cba77..1efe9b1 100644 --- a/kv-service/server/src/io_executor/tier_b.rs +++ b/kv-service/server/src/io_executor/tier_b.rs @@ -23,7 +23,7 @@ use super::{ log_io_batch, log_io_error, log_io_request, AlignedBuffer, IOExecutor, IORequest, IoBatchStats, - IoLogContext, + IoLogContext, O_DIRECT_FLAG, }; use crate::error::{KVError, Result}; use crossbeam_channel as channel; @@ -42,7 +42,6 @@ const DEFAULT_QUEUE_DEPTH: u32 = 256; /// O_DIRECT / 4KB alignment — must match tier_a's AlignedBuffer value. /// (Copied from tier_a::DIRECT_IO_ALIGN; duplicated here to avoid a cross-module pub.) -const O_DIRECT_FLAG: i32 = 0o40000; const DIRECT_IO_ALIGN: usize = 4096; /// Per-device worker. @@ -485,8 +484,7 @@ fn ring_worker_loop(device_idx: usize, rx: channel::Receiver, queue_dep }); let batch_len = batch.len(); let requested_bytes: usize = reqs.iter().map(|r| r.2).sum(); - let completed_bytes: usize = - ok_bytes_per_req.iter().filter_map(|x| *x).sum(); + let completed_bytes: usize = ok_bytes_per_req.iter().filter_map(|x| *x).sum(); let success_count = ok_bytes_per_req.iter().filter(|r| r.is_some()).count(); let context = IoLogContext { executor: "tier_b", @@ -1336,8 +1334,7 @@ fn do_read_aligned_into_ptr_batch_incremental( results[i] = Ok(0); continue; } - let aligned_len = - (requested_len + DIRECT_IO_ALIGN - 1) & !(DIRECT_IO_ALIGN - 1); + let aligned_len = (requested_len + DIRECT_IO_ALIGN - 1) & !(DIRECT_IO_ALIGN - 1); if capacity < aligned_len { results[i] = Err(KVError::Internal(format!( "read_aligned_into_ptr: capacity {} < aligned requested range {} \ @@ -2281,6 +2278,11 @@ mod tests { use super::*; use tempfile::TempDir; + #[test] + fn direct_io_flag_matches_native_platform() { + assert_eq!(O_DIRECT_FLAG, libc::O_DIRECT); + } + fn setup_executor(tmp: &TempDir, n_devices: usize) -> TierBExecutor { let mut exec = TierBExecutor::new(64, n_devices).unwrap(); for i in 0..n_devices { @@ -2353,7 +2355,9 @@ mod tests { .iter() .enumerate() .map(|(index, payload)| { - let path = tmp.path().join(format!("nvme{}/stripe-{}.bin", index % 2, index)); + let path = tmp + .path() + .join(format!("nvme{}/stripe-{}.bin", index % 2, index)); std::fs::write(&path, payload).unwrap(); path }) @@ -2386,9 +2390,8 @@ mod tests { assert_eq!(bytes_read.unwrap(), DIRECT_IO_ALIGN); assert!(!seen[index], "duplicate completion for stripe {}", index); seen[index] = true; - let actual = unsafe { - std::slice::from_raw_parts(buffers[index].as_mut_ptr(), DIRECT_IO_ALIGN) - }; + let actual = + unsafe { std::slice::from_raw_parts(buffers[index].as_mut_ptr(), DIRECT_IO_ALIGN) }; assert_eq!(actual, payloads[index].as_slice()); } assert_eq!(seen, vec![true; payloads.len()]); diff --git a/kv-service/server/src/rdma/wire.rs b/kv-service/server/src/rdma/wire.rs index 82ab256..214fd29 100644 --- a/kv-service/server/src/rdma/wire.rs +++ b/kv-service/server/src/rdma/wire.rs @@ -700,7 +700,6 @@ pub fn recv_put_stripes_resp(stream: &mut TcpStream) -> Result Date: Fri, 2 Oct 2026 01:06:46 +0800 Subject: [PATCH 03/20] Keep multi-rail staging live through publish and honor SGE on fallback Retain active-read and staging reservations across the post-transfer metadata lookup, serialize cancellation with final publication, and route slab-fallback writes through tag-15 destination segments. Keep uncertain server WRITE sources registered until their QP is destroyed. Add focused regression tests and document checksum configuration limits. --- README.md | 17 +++- kv-service/client-rs/src/lib.rs | 11 +- kv-service/client-rs/src/rail_read.rs | 88 ++++++++++++---- kv-service/client-rs/src/rail_read/tests.rs | 74 +++++++++++++- kv-service/server/src/rdma/qp.rs | 19 +++- kv-service/server/src/rdma/server.rs | 105 ++++++++++++++++---- 6 files changed, 268 insertions(+), 46 deletions(-) diff --git a/README.md b/README.md index 0cd13fb..091c6a5 100644 --- a/README.md +++ b/README.md @@ -335,10 +335,19 @@ every returned object. Set the server's cache policy and disk-read forcing identically for both runs; these flags cannot prove a network bottleneck by themselves. The SDK's `KvClient::read_multi_rail_into` rechecks descriptor and placement identity after the transfer and publishes bytes only after all -rails and checksums succeed. A failed or cancelled read leaves its caller -buffer unchanged. Each rail owns a compact registered receive buffer, and -the reader bounds active requests, in-flight bytes, staging, and registered -memory. No in-request transparent retry is attempted. +rails and available checksums succeed. To enforce per-stripe checksum +validation, enable `verify_stripe_checksums` on the server and rewrite the +objects being tested so their placements contain checksums. With the server's +default setting, older placements may have no checksums. Cancellation and +publishing share a gate, so whichever starts first determines the result; +a failed or cancelled read +leaves its caller buffer unchanged. Each rail owns a compact registered +receive buffer. The reader retains the active-read and final-staging budget +through the post-read lookup and publish step; transfer and MR reservations +end after all rail workers finish. No in-request transparent retry is attempted. +The server's stripe-subset fallback also honors tag-15 scatter destinations +when its registered slab cannot provide staging space. A fallback WRITE with +uncertain completion retains its source and MR until its QP is destroyed. For hardware-independent scheduling and failure checks, run `cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma diff --git a/kv-service/client-rs/src/lib.rs b/kv-service/client-rs/src/lib.rs index b5b0dda..81aeaca 100644 --- a/kv-service/client-rs/src/lib.rs +++ b/kv-service/client-rs/src/lib.rs @@ -251,15 +251,22 @@ impl KvClient { .clone() .ok_or_else(|| anyhow::anyhow!("lookup returned no RDMA placement"))?; let descriptor = initial.descriptor.clone(); + let fetch_cancel = cancel.clone(); let payload = tokio::task::spawn_blocking(move || { - reader.read_staged(&descriptor, &placement, cancel.as_ref()) + reader.read_staged(&descriptor, &placement, fetch_cancel.as_ref()) }) .await??; let current = self .lookup_object(namespace, object_key) .await? .ok_or(rail_read::RailReadError::VersionChanged)?; - let copied = rail_read::commit_if_unchanged(&initial, ¤t, &payload, destination)?; + let copied = rail_read::commit_if_unchanged( + &initial, + ¤t, + payload.as_bytes(), + destination, + cancel.as_ref(), + )?; Ok(Some(copied)) } diff --git a/kv-service/client-rs/src/rail_read.rs b/kv-service/client-rs/src/rail_read.rs index 8f19332..561efb3 100644 --- a/kv-service/client-rs/src/rail_read.rs +++ b/kv-service/client-rs/src/rail_read.rs @@ -180,6 +180,7 @@ pub(crate) fn commit_if_unchanged( current: &crate::ObjectLookup, payload: &[u8], destination: &mut [u8], + cancel: Option<&RailCancel>, ) -> Result { if !same_descriptor_identity(&initial.descriptor, ¤t.descriptor) || initial.placement != current.placement @@ -200,23 +201,42 @@ pub(crate) fn commit_if_unchanged( have: destination.len(), }); } - destination[..size].copy_from_slice(payload); + if let Some(cancel) = cancel { + cancel.publish_if_live(|| destination[..size].copy_from_slice(payload))?; + } else { + destination[..size].copy_from_slice(payload); + } Ok(size) } /// Cooperative request cancellation shared with the caller. +#[derive(Default)] +struct CancelState { + cancelled: AtomicBool, + publish_gate: Mutex<()>, +} + #[derive(Clone, Default)] -pub struct RailCancel(Arc); +pub struct RailCancel(Arc); impl RailCancel { /// Request cancellation; already started rail workers still quiesce. pub fn cancel(&self) { - self.0.store(true, Ordering::Release); + let _gate = self.0.publish_gate.lock().unwrap(); + self.0.cancelled.store(true, Ordering::Release); } /// Whether cancellation has been requested. pub fn is_cancelled(&self) -> bool { - self.0.load(Ordering::Acquire) + self.0.cancelled.load(Ordering::Acquire) + } + + fn publish_if_live(&self, publish: impl FnOnce() -> T) -> Result { + let _gate = self.0.publish_gate.lock().unwrap(); + if self.is_cancelled() { + return Err(RailReadError::Cancelled); + } + Ok(publish()) } } @@ -515,19 +535,31 @@ pub struct RailReader { topologies: Vec, limits: RailLimits, counters: Vec, - budget: Mutex, + budget: Arc>, } -struct BudgetGuard<'a> { - reader: &'a RailReader, +struct BudgetGuard { + state: Arc>, staging: u64, registered: u64, inflight: u64, } -impl Drop for BudgetGuard<'_> { +impl BudgetGuard { + fn finish_transfer(&mut self) { + let mut budget = self.state.lock().unwrap(); + budget.staging_bytes -= self.registered; + budget.registered_bytes -= self.registered; + budget.inflight_bytes -= self.inflight; + self.staging -= self.registered; + self.registered = 0; + self.inflight = 0; + } +} + +impl Drop for BudgetGuard { fn drop(&mut self) { - let mut budget = self.reader.budget.lock().unwrap(); + let mut budget = self.state.lock().unwrap(); budget.active_reads -= 1; budget.staging_bytes -= self.staging; budget.registered_bytes -= self.registered; @@ -535,6 +567,20 @@ impl Drop for BudgetGuard<'_> { } } +/// A verified private payload that retains its staging and active-read budget +/// until the caller publishes or discards the bytes. +pub struct StagedRailRead { + bytes: Vec, + _budget: BudgetGuard, +} + +impl StagedRailRead { + /// Complete object bytes in descriptor order. + pub fn as_bytes(&self) -> &[u8] { + &self.bytes + } +} + impl RailReader { /// Validate routes and create an RDMA reader without opening connections. pub fn new(routes: Vec, limits: RailLimits) -> Result { @@ -585,7 +631,7 @@ impl RailReader { topologies, limits, counters, - budget: Mutex::new(BudgetState::default()), + budget: Arc::new(Mutex::new(BudgetState::default())), }) } @@ -628,7 +674,7 @@ impl RailReader { true } - fn reserve(&self, plan: &RailPlan) -> Result, RailReadError> { + fn reserve(&self, plan: &RailPlan) -> Result { let registered = plan .tasks .iter() @@ -666,7 +712,7 @@ impl RailReader { budget.registered_bytes += registered; budget.inflight_bytes += inflight; Ok(BudgetGuard { - reader: self, + state: Arc::clone(&self.budget), staging, registered, inflight, @@ -679,13 +725,13 @@ impl RailReader { placement: &pb::PlacementDescriptor, transport: &T, cancel: Option<&RailCancel>, - ) -> Result, RailReadError> { + ) -> Result { let mut available = self.routes.clone(); for (route, counters) in available.iter_mut().zip(&self.counters) { route.enabled = counters.is_available(); } let plan = RailPlan::build(descriptor, placement, &available)?; - let _budget = self.reserve(&plan)?; + let mut budget = self.reserve(&plan)?; if cancel.is_some_and(RailCancel::is_cancelled) { return Err(RailReadError::Cancelled); } @@ -771,7 +817,11 @@ impl RailReader { }); } } - Ok(staged) + budget.finish_transfer(); + Ok(StagedRailRead { + bytes: staged, + _budget: budget, + }) } fn read_into_with( @@ -792,7 +842,11 @@ impl RailReader { }); } let staged = self.read_staged_with(descriptor, placement, transport, cancel)?; - destination[..size].copy_from_slice(&staged); + if let Some(cancel) = cancel { + cancel.publish_if_live(|| destination[..size].copy_from_slice(staged.as_bytes()))?; + } else { + destination[..size].copy_from_slice(staged.as_bytes()); + } Ok(size) } @@ -804,7 +858,7 @@ impl RailReader { descriptor: &pb::ObjectDescriptor, placement: &pb::PlacementDescriptor, cancel: Option<&RailCancel>, - ) -> Result, RailReadError> { + ) -> Result { self.read_staged_with(descriptor, placement, &VerbsTransport, cancel) } diff --git a/kv-service/client-rs/src/rail_read/tests.rs b/kv-service/client-rs/src/rail_read/tests.rs index e6e2851..91824e6 100644 --- a/kv-service/client-rs/src/rail_read/tests.rs +++ b/kv-service/client-rs/src/rail_read/tests.rs @@ -165,6 +165,74 @@ fn registration_budget_rejects_before_transport() { assert!(mock.calls.lock().unwrap().is_empty()); } +#[test] +fn completed_payload_keeps_staging_and_active_read_reserved_until_drop() { + let mock = MockTransport::new(vec![0x42; 64]); + let limits = RailLimits { + max_active_reads: 1, + ..RailLimits::default() + }; + let reader = RailReader::new(routes(), limits).unwrap(); + let (descriptor, placement) = fixture(64, 8); + let staged = reader + .read_staged_with(&descriptor, &placement, &mock, None) + .unwrap(); + assert_eq!(reader.budget.lock().unwrap().active_reads, 1); + assert!(reader.budget.lock().unwrap().staging_bytes >= 64); + assert!(matches!( + reader.read_staged_with(&descriptor, &placement, &mock, None), + Err(RailReadError::ResourceExhausted(_)) + )); + drop(staged); + assert_eq!(reader.budget.lock().unwrap().active_reads, 0); +} + +#[test] +fn cancellation_before_publish_preserves_caller_buffer() { + let (descriptor, placement) = fixture(64, 8); + let lookup = crate::ObjectLookup { + descriptor, + placement: Some(placement), + }; + let cancel = RailCancel::default(); + cancel.cancel(); + let mut destination = vec![0xA5; 64]; + assert!(matches!( + commit_if_unchanged( + &lookup, + &lookup, + &[0x42; 64], + &mut destination, + Some(&cancel) + ), + Err(RailReadError::Cancelled) + )); + assert_eq!(destination, vec![0xA5; 64]); +} + +#[test] +fn cancellation_and_publish_have_one_order() { + use std::sync::mpsc; + let token = RailCancel::default(); + let (publishing_tx, publishing_rx) = mpsc::channel(); + let (release_tx, release_rx) = mpsc::channel(); + let committing = token.clone(); + let publish = std::thread::spawn(move || { + committing.publish_if_live(|| { + publishing_tx.send(()).unwrap(); + release_rx.recv().unwrap(); + }) + }); + publishing_rx.recv().unwrap(); + let cancelling = token.clone(); + let cancel = std::thread::spawn(move || cancelling.cancel()); + assert!(!token.is_cancelled()); + release_tx.send(()).unwrap(); + publish.join().unwrap().unwrap(); + cancel.join().unwrap(); + assert!(token.is_cancelled()); +} + #[test] fn descriptor_identity_rejects_generation_and_layout_changes() { let (original, _) = fixture(64, 8); @@ -502,20 +570,20 @@ fn changed_generation_cannot_publish_completed_rail_bytes() { current.descriptor.object_generation += 1; let mut destination = vec![0xA5; 64]; assert!(matches!( - commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination), + commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination, None), Err(RailReadError::VersionChanged) )); assert_eq!(destination, vec![0xA5; 64]); current = initial.clone(); current.placement.as_mut().unwrap().layout_hash = "moved".into(); assert!(matches!( - commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination), + commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination, None), Err(RailReadError::VersionChanged) )); assert_eq!(destination, vec![0xA5; 64]); current = initial.clone(); assert_eq!( - commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination).unwrap(), + commit_if_unchanged(&initial, ¤t, &[0x42; 64], &mut destination, None).unwrap(), 64 ); assert_eq!(destination, vec![0x42; 64]); diff --git a/kv-service/server/src/rdma/qp.rs b/kv-service/server/src/rdma/qp.rs index 6f758af..69af176 100644 --- a/kv-service/server/src/rdma/qp.rs +++ b/kv-service/server/src/rdma/qp.rs @@ -1,9 +1,11 @@ //! Reliable Connection QP — one-to-one connection with a client; use ibv_post_send WRITE after handshake -use crate::rdma::context::RdmaContext; +use crate::rdma::context::{MemRegion, RdmaContext}; use anyhow::{anyhow, Result}; +use prost::bytes::Bytes; use rdma_sys::*; use std::ptr::{self, NonNull}; +use std::sync::Mutex; /// An RC QP (Reliable Connection Queue Pair). /// @@ -15,6 +17,9 @@ use std::ptr::{self, NonNull}; /// 5. `qp.post_write(...)` — actual operation pub struct RcQp { qp: NonNull, + // A timed-out WRITE may still read its local MR. These buffers are + // released only after Drop destroys the QP. + retired_writes: Mutex>, /// Local QP info, sent to remote over the control plane pub local: QpInfo, } @@ -105,7 +110,11 @@ impl RcQp { tracing::info!("RcQp created: qpn={} psn=0x{:x}", local.qpn, local.psn); - Ok(Self { qp, local }) + Ok(Self { + qp, + retired_writes: Mutex::new(Vec::new()), + local, + }) } } @@ -310,6 +319,12 @@ impl RcQp { Ok(n as usize) } } + + /// Retain an uncertain local WRITE source until this QP is destroyed. + /// Used when polling cannot prove a signaled WRITE has quiesced. + pub fn retain_uncertain_write(&self, mr: MemRegion, source: Bytes) { + self.retired_writes.lock().unwrap().push((mr, source)); + } } fn wc_status_str(status: u32) -> &'static str { diff --git a/kv-service/server/src/rdma/server.rs b/kv-service/server/src/rdma/server.rs index ef6801b..cfce624 100644 --- a/kv-service/server/src/rdma/server.rs +++ b/kv-service/server/src/rdma/server.rs @@ -1089,6 +1089,27 @@ fn map_range_to_segments( Ok(out) } +fn fallback_write_targets( + segments: &[(u64, u32, u64)], + dst_addr: u64, + dst_rkey: u32, + stripe_index: usize, + chunk_size: u64, + length: usize, +) -> Result> { + let offset = (stripe_index as u64) + .checked_mul(chunk_size) + .ok_or_else(|| anyhow!("stripe destination offset overflow"))?; + if segments.is_empty() { + let addr = dst_addr + .checked_add(offset) + .ok_or_else(|| anyhow!("stripe destination address overflow"))?; + Ok(vec![(addr, dst_rkey, length as u64)]) + } else { + map_range_to_segments(segments, offset, length as u64) + } +} + fn serve_get_stripes_fallback( kv_ctx: &Arc, rdma: &Arc, @@ -1111,11 +1132,20 @@ fn serve_get_stripes_fallback( }; let mut total = 0u64; - let mut mrs = Vec::with_capacity(segments.len()); - for (write_index, (stripe_index, segment)) in segments.iter().enumerate() { + let chunk_count = segments.len() as u32; + let mut write_index = 0u64; + for (stripe_index, segment) in segments { if segment.is_empty() { continue; } + let targets = fallback_write_targets( + &req.dst_segments, + req.dst_addr, + req.dst_rkey, + stripe_index, + striping.chunk_size, + segment.len(), + )?; let mr = unsafe { rdma.register_mr_raw( segment.as_ptr() as *mut u8, @@ -1123,22 +1153,29 @@ fn serve_get_stripes_fallback( ibv_access_flags::IBV_ACCESS_LOCAL_WRITE.0, )? }; - qp.post_write( - write_index as u64, - mr.addr, - mr.lkey, - req.dst_addr + *stripe_index as u64 * striping.chunk_size, - req.dst_rkey, - segment.len() as u32, - write_index + 1 == segments.len(), - )?; + let mut source_addr = mr.addr; + for (target_addr, target_rkey, length) in targets { + let length = u32::try_from(length) + .map_err(|_| anyhow!("fallback RDMA WRITE length exceeds u32"))?; + qp.post_write( + write_index, + source_addr, + mr.lkey, + target_addr, + target_rkey, + length, + true, + )?; + if let Err(error) = RcQp::poll_n(client_cq, 1) { + qp.retain_uncertain_write(mr, segment); + return Err(error); + } + source_addr += u64::from(length); + write_index += 1; + } total += segment.len() as u64; - mrs.push(mr); - } - if !mrs.is_empty() { - RcQp::poll_n(client_cq, 1)?; } - Ok((true, total, segments.len() as u32)) + Ok((true, total, chunk_count)) } /// Serve a stripe-subset descriptor GET (tag 12): read each requested stripe @@ -1204,8 +1241,22 @@ fn serve_get_stripes( return Ok((false, 0, 0)); } let stripe_offset = idx as u64 * chunk_size; + let stripe_length = chunk_size.min(striping.total_size.saturating_sub(stripe_offset)); + if fallback_write_targets( + &req.dst_segments, + req.dst_addr, + req.dst_rkey, + idx, + chunk_size, + stripe_length as usize, + ) + .is_err() + { + tracing::warn!("stripe-subset GET destination does not cover stripe {}", idx); + return Ok((false, 0, 0)); + } staged_bytes = staged_bytes - .checked_add(chunk_size.min(striping.total_size.saturating_sub(stripe_offset)) as usize) + .checked_add(stripe_length as usize) .ok_or_else(|| anyhow!("stripe subset staging size overflow"))?; } @@ -2011,7 +2062,7 @@ mod tests { #[cfg(test)] mod sge_tests { - use super::map_range_to_segments; + use super::{fallback_write_targets, map_range_to_segments}; #[test] fn range_within_one_segment() { @@ -2039,4 +2090,22 @@ mod sge_tests { let m = map_range_to_segments(&segs, 64, 64).unwrap(); assert_eq!(m, vec![(0x9000, 8, 64)]); } + + #[test] + fn fallback_targets_use_compact_sge_addresses_for_noncontiguous_stripes() { + let segs = [ + (0x1000, 11, 64), + (0x9000, 12, 64), + (0x1040, 11, 64), + (0x9000, 12, 64), + ]; + assert_eq!( + fallback_write_targets(&segs, 0x1000, 11, 2, 64, 64).unwrap(), + vec![(0x1040, 11, 64)] + ); + assert_eq!( + fallback_write_targets(&segs, 0x1000, 11, 3, 64, 64).unwrap(), + vec![(0x9000, 12, 64)] + ); + } } From 6c805e01d84c88b53508e7c1f35fe8714b8cac91 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 01:10:22 +0800 Subject: [PATCH 04/20] Record reproducible software-only rail benchmark samples Add a parameterized collector for paired one- and two-rail Mock reads, and commit the raw 180 latency samples, run summaries, and environment manifest. Label the data as software-only so it cannot be mistaken for HCA throughput. --- README.md | 6 +- kv-service/benchmarks/collect_rail_mock.py | 144 ++++++++++++++ ...-02-independent-rail-mock-environment.json | 21 ++ ...26-10-02-independent-rail-mock-samples.csv | 181 ++++++++++++++++++ ...26-10-02-independent-rail-mock-summary.csv | 19 ++ 5 files changed, 370 insertions(+), 1 deletion(-) create mode 100644 kv-service/benchmarks/collect_rail_mock.py create mode 100644 kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json create mode 100644 kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv create mode 100644 kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv diff --git a/README.md b/README.md index 091c6a5..412baee 100644 --- a/README.md +++ b/README.md @@ -354,7 +354,11 @@ For hardware-independent scheduling and failure checks, run rail_read::tests`. The ignored `rail_read_e2e` tests require two reachable RDMA listeners and the `CS_RAIL_*` endpoint/device environment variables. The ignored `software_only_mock_benchmark` exercises scheduling and memory -copies; its throughput is **not** an RDMA hardware result. +copies; its throughput is **not** an RDMA hardware result. Reproduce its +paired 1/2-rail matrix and save all per-read samples with +`python kv-service/benchmarks/collect_rail_mock.py --sizes 64,256,512`. +The checked-in `kv-service/benchmarks/results/2026-10-02-independent-rail-mock-*` +files contain one Linux ARM64 software-only run, including environment details. --- diff --git a/kv-service/benchmarks/collect_rail_mock.py b/kv-service/benchmarks/collect_rail_mock.py new file mode 100644 index 0000000..6bfa1f6 --- /dev/null +++ b/kv-service/benchmarks/collect_rail_mock.py @@ -0,0 +1,144 @@ +from __future__ import annotations + +# Collect software-only rail scheduling samples from the Rust Mock test. +# Example: python kv-service/benchmarks/collect_rail_mock.py --sizes 64,256,512 +# This measures memory copies and scheduling, not RDMA or disk bandwidth. + +import argparse +import csv +import json +import os +import platform +import re +import subprocess +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[2] +MANIFEST = ROOT / "kv-service/client-rs/Cargo.toml" +SAMPLES = re.compile(r"mock_samples_us=\[([^]]+)\]") +SUMMARY = re.compile( + r"mock_only,rails=(\d+),size_bytes=(\d+),iters=(\d+)," + r"avg_us=(\d+),gib_per_s=([0-9.]+),cpu_user_us=(\d+)," + r"cpu_system_us=(\d+),peak_rss_kb=(\d+),rail_bytes=\[([^]]+)\]" +) + + +def read_command(*args: str) -> str: + return subprocess.check_output(args, cwd=ROOT, text=True).strip() + + +def write_csv(path: Path, rows: list[dict[str, int | float]]) -> None: + with path.open("w", newline="") as handle: + writer = csv.DictWriter(handle, fieldnames=rows[0].keys()) + writer.writeheader() + writer.writerows(rows) + + +def main() -> None: + parser = argparse.ArgumentParser(description="Collect software-only rail Mock samples") + parser.add_argument("--sizes", default="64,256,512", help="Comma-separated MiB sizes") + parser.add_argument("--trials", type=int, default=3) + parser.add_argument("--iterations", type=int, default=10) + parser.add_argument("--output-prefix", default="rail-mock") + parser.add_argument("--environment-note", default="") + parser.add_argument( + "--source-commit", help="Source commit when running a copied tree without .git" + ) + args = parser.parse_args() + sizes = [int(value) for value in args.sizes.split(",")] + if not sizes or any(value <= 0 for value in sizes): + parser.error("--sizes must contain positive MiB values") + if args.trials <= 0 or args.iterations <= 0: + parser.error("--trials and --iterations must be positive") + + summary_rows: list[dict[str, int | float]] = [] + sample_rows: list[dict[str, int]] = [] + for trial in range(1, args.trials + 1): + for size in sizes: + for rails in (1, 2): + env = os.environ.copy() + env.update( + CS_RAIL_MOCK_MIB=str(size), + CS_RAIL_MOCK_ITERS=str(args.iterations), + CS_RAIL_MOCK_RAILS=str(rails), + ) + command = [ + "cargo", + "test", + "--manifest-path", + str(MANIFEST), + "--release", + "--features", + "rdma", + "--lib", + "software_only_mock_benchmark", + "--", + "--ignored", + "--nocapture", + ] + output = subprocess.run( + command, + cwd=ROOT, + env=env, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + check=True, + ).stdout + samples = SAMPLES.search(output) + summary = SUMMARY.search(output) + if samples is None or summary is None: + raise RuntimeError(f"Mock benchmark output lacks samples: {output[-2000:]}") + values = [int(value) for value in samples.group(1).split(",")] + _, size_bytes, iterations, avg_us, gib, user_us, system_us, rss_kb, rail_text = ( + summary.groups() + ) + rail_bytes = [int(value) for value in rail_text.split(",")] + if len(values) != args.iterations or int(size_bytes) != size * 1024 * 1024: + raise RuntimeError("Mock benchmark sample count or object size disagrees") + summary_rows.append( + dict( + trial=trial, + size_mib=size, + rails=rails, + iterations=int(iterations), + mean_read_us=int(avg_us), + effective_gib_s=float(gib), + cpu_user_us=int(user_us), + cpu_system_us=int(system_us), + peak_rss_kb=int(rss_kb), + rail0_bytes=rail_bytes[0], + rail1_bytes=rail_bytes[1] if rails == 2 else 0, + ) + ) + sample_rows.extend( + dict(trial=trial, size_mib=size, rails=rails, iteration=index, read_us=value) + for index, value in enumerate(values, 1) + ) + print(f"trial={trial} size={size}MiB rails={rails} avg_us={avg_us}", flush=True) + + output_dir = Path(__file__).resolve().parent / "results" + output_dir.mkdir(exist_ok=True) + write_csv(output_dir / f"{args.output_prefix}-summary.csv", summary_rows) + write_csv(output_dir / f"{args.output_prefix}-samples.csv", sample_rows) + metadata = { + "environment": "Mock software-only; no RDMA Verbs or disk I/O", + "environment_note": args.environment_note, + "git_commit": args.source_commit or read_command("git", "rev-parse", "HEAD"), + "platform": platform.platform(), + "machine": platform.machine(), + "rustc": read_command("rustc", "--version"), + "sizes_mib": sizes, + "trials": args.trials, + "iterations": args.iterations, + "stripe_mib": 4, + "warmup_reads": 1, + } + (output_dir / f"{args.output_prefix}-environment.json").write_text( + json.dumps(metadata, indent=2) + "\n" + ) + + +if __name__ == "__main__": + main() diff --git a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json new file mode 100644 index 0000000..6bcfc8c --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json @@ -0,0 +1,21 @@ +{ + "environment": "Mock software-only; no RDMA Verbs, network, or disk I/O", + "host_cpu": "Apple M5 Pro", + "guest": "Debian GNU/Linux 12, Linux aarch64, native OrbStack container", + "rustc": "1.94.1 (e408947bf 2026-03-25)", + "git_commit": "c0543f4", + "object_sizes_mib": [ + 64, + 256, + 512 + ], + "stripe_mib": 4, + "warmup_reads_per_run": 1, + "independent_trials": 3, + "iterations_per_trial": 10, + "rail_counts": [ + 1, + 2 + ], + "notes": "One rail and two rail runs use the same in-memory object, chunk layout, client, and host. Rows are paired in trial/size order. Per-rail bytes were equal in two-rail runs. This is not hardware RDMA evidence." +} diff --git a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv new file mode 100644 index 0000000..c3a26d4 --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv @@ -0,0 +1,181 @@ +trial,size_mib,rails,iteration,read_us +1,64,1,1,11386 +1,64,1,2,10548 +1,64,1,3,10426 +1,64,1,4,10267 +1,64,1,5,9858 +1,64,1,6,9869 +1,64,1,7,10073 +1,64,1,8,9534 +1,64,1,9,9549 +1,64,1,10,9499 +1,64,2,1,8300 +1,64,2,2,8096 +1,64,2,3,8590 +1,64,2,4,8737 +1,64,2,5,8403 +1,64,2,6,8687 +1,64,2,7,8045 +1,64,2,8,7939 +1,64,2,9,7903 +1,64,2,10,8061 +1,256,1,1,39301 +1,256,1,2,36680 +1,256,1,3,37680 +1,256,1,4,36160 +1,256,1,5,36215 +1,256,1,6,36377 +1,256,1,7,36038 +1,256,1,8,35714 +1,256,1,9,35789 +1,256,1,10,35305 +1,256,2,1,31837 +1,256,2,2,31642 +1,256,2,3,32240 +1,256,2,4,32086 +1,256,2,5,31821 +1,256,2,6,32547 +1,256,2,7,31964 +1,256,2,8,32157 +1,256,2,9,32247 +1,256,2,10,31257 +1,512,1,1,81294 +1,512,1,2,70710 +1,512,1,3,69650 +1,512,1,4,68923 +1,512,1,5,69410 +1,512,1,6,69077 +1,512,1,7,72094 +1,512,1,8,69755 +1,512,1,9,67387 +1,512,1,10,67667 +1,512,2,1,61260 +1,512,2,2,60351 +1,512,2,3,60140 +1,512,2,4,63156 +1,512,2,5,64755 +1,512,2,6,61920 +1,512,2,7,61791 +1,512,2,8,62816 +1,512,2,9,62670 +1,512,2,10,60840 +2,64,1,1,9048 +2,64,1,2,9102 +2,64,1,3,9387 +2,64,1,4,9469 +2,64,1,5,9423 +2,64,1,6,9692 +2,64,1,7,9428 +2,64,1,8,9237 +2,64,1,9,9328 +2,64,1,10,9192 +2,64,2,1,8534 +2,64,2,2,8504 +2,64,2,3,8426 +2,64,2,4,8465 +2,64,2,5,8479 +2,64,2,6,8553 +2,64,2,7,8379 +2,64,2,8,8475 +2,64,2,9,8227 +2,64,2,10,8469 +2,256,1,1,35927 +2,256,1,2,37936 +2,256,1,3,36306 +2,256,1,4,36228 +2,256,1,5,36804 +2,256,1,6,35403 +2,256,1,7,35714 +2,256,1,8,36027 +2,256,1,9,36621 +2,256,1,10,36597 +2,256,2,1,31978 +2,256,2,2,31533 +2,256,2,3,31932 +2,256,2,4,31528 +2,256,2,5,31487 +2,256,2,6,31930 +2,256,2,7,33285 +2,256,2,8,31466 +2,256,2,9,32205 +2,256,2,10,31718 +2,512,1,1,66914 +2,512,1,2,68774 +2,512,1,3,68994 +2,512,1,4,67744 +2,512,1,5,67523 +2,512,1,6,69898 +2,512,1,7,69765 +2,512,1,8,69054 +2,512,1,9,68442 +2,512,1,10,68989 +2,512,2,1,60857 +2,512,2,2,60558 +2,512,2,3,62415 +2,512,2,4,61730 +2,512,2,5,62048 +2,512,2,6,61636 +2,512,2,7,61733 +2,512,2,8,61449 +2,512,2,9,61961 +2,512,2,10,62689 +3,64,1,1,9760 +3,64,1,2,9844 +3,64,1,3,9933 +3,64,1,4,9870 +3,64,1,5,9545 +3,64,1,6,9697 +3,64,1,7,9589 +3,64,1,8,9887 +3,64,1,9,9801 +3,64,1,10,9967 +3,64,2,1,8410 +3,64,2,2,8477 +3,64,2,3,8650 +3,64,2,4,11078 +3,64,2,5,16520 +3,64,2,6,12802 +3,64,2,7,12738 +3,64,2,8,8634 +3,64,2,9,8516 +3,64,2,10,8259 +3,256,1,1,36533 +3,256,1,2,36790 +3,256,1,3,36643 +3,256,1,4,35873 +3,256,1,5,38027 +3,256,1,6,36220 +3,256,1,7,36321 +3,256,1,8,36267 +3,256,1,9,36183 +3,256,1,10,35812 +3,256,2,1,31827 +3,256,2,2,32730 +3,256,2,3,32601 +3,256,2,4,33006 +3,256,2,5,32154 +3,256,2,6,32364 +3,256,2,7,33601 +3,256,2,8,32121 +3,256,2,9,32159 +3,256,2,10,32311 +3,512,1,1,70976 +3,512,1,2,68575 +3,512,1,3,70807 +3,512,1,4,69899 +3,512,1,5,82374 +3,512,1,6,67953 +3,512,1,7,70938 +3,512,1,8,69135 +3,512,1,9,69190 +3,512,1,10,69697 +3,512,2,1,61740 +3,512,2,2,62476 +3,512,2,3,60883 +3,512,2,4,60562 +3,512,2,5,62436 +3,512,2,6,61313 +3,512,2,7,60699 +3,512,2,8,63034 +3,512,2,9,62417 +3,512,2,10,60530 diff --git a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv new file mode 100644 index 0000000..85860b5 --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv @@ -0,0 +1,19 @@ +trial,size_mib,rails,iterations,mean_read_us,effective_gib_s,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes +1,64,1,10,10100,6.188,40708,56027,264504,671088640,0 +1,64,2,10,8276,7.552,39121,53618,264336,335544320,335544320 +1,256,1,10,36525,6.845,179122,180736,1050868,2684354560,0 +1,256,2,10,31979,7.818,185550,215482,1050864,1342177280,1342177280 +1,512,1,10,70596,7.083,353435,347682,2099584,5368709120,0 +1,512,2,10,61969,8.069,385636,407558,2099568,2684354560,2684354560 +2,64,1,10,9330,6.699,38601,50319,264504,671088640,0 +2,64,2,10,8451,7.396,46332,51551,264328,335544320,335544320 +2,256,1,10,36356,6.876,183907,175276,1050988,2684354560,0 +2,256,2,10,31906,7.836,185868,213359,1050856,1342177280,1342177280 +2,512,1,10,68609,7.288,360518,320170,2099600,5368709120,0 +2,512,2,10,61707,8.103,391285,394308,2099472,2684354560,2684354560 +3,64,1,10,9789,6.385,36027,56529,264504,671088640,0 +3,64,2,10,10408,6.005,51706,71837,264324,335544320,335544320 +3,256,1,10,36466,6.856,164191,196881,1050848,2684354560,0 +3,256,2,10,32487,7.695,189599,213688,1050876,1342177280,1342177280 +3,512,1,10,70954,7.047,376364,328773,2099592,5368709120,0 +3,512,2,10,61609,8.116,395817,387664,2099552,2684354560,2684354560 From fe778d0516c857ff54baf1ae56123345740e8fdc Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 01:16:25 +0800 Subject: [PATCH 05/20] Bound aggregate in-flight work on each RDMA rail Reserve per-rail task slots and bytes atomically across concurrent requests, release the transfer credits when workers finish, and retain the final-object budget through publication. Add red-green regression tests for both per-rail limits. --- README.md | 8 +-- kv-service/client-rs/src/rail_read.rs | 52 ++++++++++++++++---- kv-service/client-rs/src/rail_read/tests.rs | 54 +++++++++++++++++++++ 3 files changed, 101 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 412baee..88b0c01 100644 --- a/README.md +++ b/README.md @@ -342,9 +342,11 @@ default setting, older placements may have no checksums. Cancellation and publishing share a gate, so whichever starts first determines the result; a failed or cancelled read leaves its caller buffer unchanged. Each rail owns a compact registered -receive buffer. The reader retains the active-read and final-staging budget -through the post-read lookup and publish step; transfer and MR reservations -end after all rail workers finish. No in-request transparent retry is attempted. +receive buffer. The reader bounds concurrent tasks and aggregate in-flight +bytes per rail across simultaneous requests. It retains the active-read and +final-staging budget through the post-read lookup and publish step; transfer +and MR reservations end after all rail workers finish. No in-request +transparent retry is attempted. The server's stripe-subset fallback also honors tag-15 scatter destinations when its registered slab cannot provide staging space. A fallback WRITE with uncertain completion retains its source and MR until its QP is destroyed. diff --git a/kv-service/client-rs/src/rail_read.rs b/kv-service/client-rs/src/rail_read.rs index 561efb3..7af7d66 100644 --- a/kv-service/client-rs/src/rail_read.rs +++ b/kv-service/client-rs/src/rail_read.rs @@ -91,13 +91,15 @@ impl RailRoute { pub struct RailLimits { /// Maximum concurrent object reads accepted by this reader. pub max_active_reads: usize, + /// Maximum concurrent transfer tasks using one configured rail. + pub max_active_reads_per_rail: usize, /// Maximum final and per-rail staging allocation reserved at once. pub max_staging_bytes: u64, /// Maximum total MR lengths reserved across active reads. pub max_registered_bytes: u64, /// Maximum object payload bytes in flight at once. pub max_inflight_bytes: u64, - /// Maximum payload bytes assigned to one rail for one request. + /// Maximum aggregate payload bytes in flight on one rail. pub max_inflight_bytes_per_rail: u64, /// TCP control deadline for connect, send, and reply operations. pub io_timeout: Duration, @@ -109,6 +111,7 @@ impl Default for RailLimits { fn default() -> Self { Self { max_active_reads: 8, + max_active_reads_per_rail: 8, max_staging_bytes: 4 * 1024 * 1024 * 1024, max_registered_bytes: 4 * 1024 * 1024 * 1024, max_inflight_bytes: 4 * 1024 * 1024 * 1024, @@ -527,6 +530,8 @@ struct BudgetState { staging_bytes: u64, registered_bytes: u64, inflight_bytes: u64, + rail_active_reads: Vec, + rail_inflight_bytes: Vec, } /// Bounded multi-rail reader; one active request uses one QP per chosen rail. @@ -543,6 +548,7 @@ struct BudgetGuard { staging: u64, registered: u64, inflight: u64, + rail_reservations: Vec<(usize, u64)>, } impl BudgetGuard { @@ -551,6 +557,10 @@ impl BudgetGuard { budget.staging_bytes -= self.registered; budget.registered_bytes -= self.registered; budget.inflight_bytes -= self.inflight; + for (index, bytes) in self.rail_reservations.drain(..) { + budget.rail_active_reads[index] -= 1; + budget.rail_inflight_bytes[index] -= bytes; + } self.staging -= self.registered; self.registered = 0; self.inflight = 0; @@ -564,6 +574,10 @@ impl Drop for BudgetGuard { budget.staging_bytes -= self.staging; budget.registered_bytes -= self.registered; budget.inflight_bytes -= self.inflight; + for (index, bytes) in &self.rail_reservations { + budget.rail_active_reads[*index] -= 1; + budget.rail_inflight_bytes[*index] -= *bytes; + } } } @@ -584,7 +598,10 @@ impl StagedRailRead { impl RailReader { /// Validate routes and create an RDMA reader without opening connections. pub fn new(routes: Vec, limits: RailLimits) -> Result { - if routes.is_empty() || limits.max_active_reads == 0 { + if routes.is_empty() + || limits.max_active_reads == 0 + || limits.max_active_reads_per_rail == 0 + { return Err(RailReadError::InvalidPlacement( "at least one rail and one active read slot are required".into(), )); @@ -626,12 +643,17 @@ impl RailReader { ..RailCounters::default() }) .collect(); + let route_count = routes.len(); Ok(Self { routes, topologies, limits, counters, - budget: Arc::new(Mutex::new(BudgetState::default())), + budget: Arc::new(Mutex::new(BudgetState { + rail_active_reads: vec![0; route_count], + rail_inflight_bytes: vec![0; route_count], + ..BudgetState::default() + })), }) } @@ -688,15 +710,11 @@ impl RailReader { .checked_add(plan.size as u64) .ok_or_else(|| RailReadError::ResourceExhausted("staging size overflow".into()))?; let inflight = plan.size as u64; - if plan + let rail_reservations: Vec<(usize, u64)> = plan .tasks .iter() - .any(|task| task.packed_len as u64 > self.limits.max_inflight_bytes_per_rail) - { - return Err(RailReadError::ResourceExhausted( - "per-rail in-flight bytes".into(), - )); - } + .map(|task| (task.route_index, task.packed_len as u64)) + .collect(); let mut budget = self.budget.lock().unwrap(); if budget.active_reads >= self.limits.max_active_reads || budget.staging_bytes.saturating_add(staging) > self.limits.max_staging_bytes @@ -707,15 +725,29 @@ impl RailReader { "active reads, staging, registration, or in-flight bytes".into(), )); } + if rail_reservations.iter().any(|(index, bytes)| { + budget.rail_active_reads[*index] >= self.limits.max_active_reads_per_rail + || budget.rail_inflight_bytes[*index].saturating_add(*bytes) + > self.limits.max_inflight_bytes_per_rail + }) { + return Err(RailReadError::ResourceExhausted( + "per-rail active tasks or aggregate in-flight bytes".into(), + )); + } budget.active_reads += 1; budget.staging_bytes += staging; budget.registered_bytes += registered; budget.inflight_bytes += inflight; + for (index, bytes) in &rail_reservations { + budget.rail_active_reads[*index] += 1; + budget.rail_inflight_bytes[*index] += *bytes; + } Ok(BudgetGuard { state: Arc::clone(&self.budget), staging, registered, inflight, + rail_reservations, }) } diff --git a/kv-service/client-rs/src/rail_read/tests.rs b/kv-service/client-rs/src/rail_read/tests.rs index 91824e6..c8b93b5 100644 --- a/kv-service/client-rs/src/rail_read/tests.rs +++ b/kv-service/client-rs/src/rail_read/tests.rs @@ -165,6 +165,46 @@ fn registration_budget_rejects_before_transport() { assert!(mock.calls.lock().unwrap().is_empty()); } +#[test] +fn per_rail_inflight_limit_counts_concurrent_requests() { + let (descriptor, placement) = fixture(64, 8); + let configured = vec![routes()[0].clone()]; + let plan = RailPlan::build(&descriptor, &placement, &configured).unwrap(); + let limits = RailLimits { + max_active_reads: 2, + max_inflight_bytes_per_rail: 96, + ..RailLimits::default() + }; + let reader = RailReader::new(configured, limits).unwrap(); + let first = reader.reserve(&plan).unwrap(); + assert!(matches!( + reader.reserve(&plan), + Err(RailReadError::ResourceExhausted(_)) + )); + drop(first); + assert!(reader.reserve(&plan).is_ok()); +} + +#[test] +fn per_rail_task_limit_counts_concurrent_requests() { + let (descriptor, placement) = fixture(64, 8); + let configured = vec![routes()[0].clone()]; + let plan = RailPlan::build(&descriptor, &placement, &configured).unwrap(); + let limits = RailLimits { + max_active_reads: 2, + max_active_reads_per_rail: 1, + ..RailLimits::default() + }; + let reader = RailReader::new(configured, limits).unwrap(); + let first = reader.reserve(&plan).unwrap(); + assert!(matches!( + reader.reserve(&plan), + Err(RailReadError::ResourceExhausted(_)) + )); + drop(first); + assert!(reader.reserve(&plan).is_ok()); +} + #[test] fn completed_payload_keeps_staging_and_active_read_reserved_until_drop() { let mock = MockTransport::new(vec![0x42; 64]); @@ -179,6 +219,20 @@ fn completed_payload_keeps_staging_and_active_read_reserved_until_drop() { .unwrap(); assert_eq!(reader.budget.lock().unwrap().active_reads, 1); assert!(reader.budget.lock().unwrap().staging_bytes >= 64); + assert!(reader + .budget + .lock() + .unwrap() + .rail_active_reads + .iter() + .all(|count| *count == 0)); + assert!(reader + .budget + .lock() + .unwrap() + .rail_inflight_bytes + .iter() + .all(|bytes| *bytes == 0)); assert!(matches!( reader.read_staged_with(&descriptor, &placement, &mock, None), Err(RailReadError::ResourceExhausted(_)) From 8c51dedec6e9b2ce90233059317fa441ee850a5e Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 01:18:12 +0800 Subject: [PATCH 06/20] Refresh rail Mock evidence after aggregate budget fix Rerun the paired 64, 256, and 512 MiB software-only matrix on the exact per-rail budget implementation. Keep the 180 raw samples and environment manifest with LF CSV output for clean review. --- kv-service/benchmarks/collect_rail_mock.py | 2 +- ...-02-independent-rail-mock-environment.json | 24 +- ...26-10-02-independent-rail-mock-samples.csv | 362 +++++++++--------- ...26-10-02-independent-rail-mock-summary.csv | 38 +- 4 files changed, 211 insertions(+), 215 deletions(-) diff --git a/kv-service/benchmarks/collect_rail_mock.py b/kv-service/benchmarks/collect_rail_mock.py index 6bfa1f6..fbaf75d 100644 --- a/kv-service/benchmarks/collect_rail_mock.py +++ b/kv-service/benchmarks/collect_rail_mock.py @@ -30,7 +30,7 @@ def read_command(*args: str) -> str: def write_csv(path: Path, rows: list[dict[str, int | float]]) -> None: with path.open("w", newline="") as handle: - writer = csv.DictWriter(handle, fieldnames=rows[0].keys()) + writer = csv.DictWriter(handle, fieldnames=rows[0].keys(), lineterminator="\n") writer.writeheader() writer.writerows(rows) diff --git a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json index 6bcfc8c..e2f3844 100644 --- a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json +++ b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-environment.json @@ -1,21 +1,17 @@ { - "environment": "Mock software-only; no RDMA Verbs, network, or disk I/O", - "host_cpu": "Apple M5 Pro", - "guest": "Debian GNU/Linux 12, Linux aarch64, native OrbStack container", - "rustc": "1.94.1 (e408947bf 2026-03-25)", - "git_commit": "c0543f4", - "object_sizes_mib": [ + "environment": "Mock software-only; no RDMA Verbs or disk I/O", + "environment_note": "Apple M5 Pro host; Debian 12 Linux aarch64 native OrbStack container", + "git_commit": "fe778d0", + "platform": "Linux-6.12.15-orbstack-00304-gd0ddcf70447d-aarch64-with-glibc2.36", + "machine": "aarch64", + "rustc": "rustc 1.94.1 (e408947bf 2026-03-25)", + "sizes_mib": [ 64, 256, 512 ], + "trials": 3, + "iterations": 10, "stripe_mib": 4, - "warmup_reads_per_run": 1, - "independent_trials": 3, - "iterations_per_trial": 10, - "rail_counts": [ - 1, - 2 - ], - "notes": "One rail and two rail runs use the same in-memory object, chunk layout, client, and host. Rows are paired in trial/size order. Per-rail bytes were equal in two-rail runs. This is not hardware RDMA evidence." + "warmup_reads": 1 } diff --git a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv index c3a26d4..03197e1 100644 --- a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv +++ b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-samples.csv @@ -1,181 +1,181 @@ -trial,size_mib,rails,iteration,read_us -1,64,1,1,11386 -1,64,1,2,10548 -1,64,1,3,10426 -1,64,1,4,10267 -1,64,1,5,9858 -1,64,1,6,9869 -1,64,1,7,10073 -1,64,1,8,9534 -1,64,1,9,9549 -1,64,1,10,9499 -1,64,2,1,8300 -1,64,2,2,8096 -1,64,2,3,8590 -1,64,2,4,8737 -1,64,2,5,8403 -1,64,2,6,8687 -1,64,2,7,8045 -1,64,2,8,7939 -1,64,2,9,7903 -1,64,2,10,8061 -1,256,1,1,39301 -1,256,1,2,36680 -1,256,1,3,37680 -1,256,1,4,36160 -1,256,1,5,36215 -1,256,1,6,36377 -1,256,1,7,36038 -1,256,1,8,35714 -1,256,1,9,35789 -1,256,1,10,35305 -1,256,2,1,31837 -1,256,2,2,31642 -1,256,2,3,32240 -1,256,2,4,32086 -1,256,2,5,31821 -1,256,2,6,32547 -1,256,2,7,31964 -1,256,2,8,32157 -1,256,2,9,32247 -1,256,2,10,31257 -1,512,1,1,81294 -1,512,1,2,70710 -1,512,1,3,69650 -1,512,1,4,68923 -1,512,1,5,69410 -1,512,1,6,69077 -1,512,1,7,72094 -1,512,1,8,69755 -1,512,1,9,67387 -1,512,1,10,67667 -1,512,2,1,61260 -1,512,2,2,60351 -1,512,2,3,60140 -1,512,2,4,63156 -1,512,2,5,64755 -1,512,2,6,61920 -1,512,2,7,61791 -1,512,2,8,62816 -1,512,2,9,62670 -1,512,2,10,60840 -2,64,1,1,9048 -2,64,1,2,9102 -2,64,1,3,9387 -2,64,1,4,9469 -2,64,1,5,9423 -2,64,1,6,9692 -2,64,1,7,9428 -2,64,1,8,9237 -2,64,1,9,9328 -2,64,1,10,9192 -2,64,2,1,8534 -2,64,2,2,8504 -2,64,2,3,8426 -2,64,2,4,8465 -2,64,2,5,8479 -2,64,2,6,8553 -2,64,2,7,8379 -2,64,2,8,8475 -2,64,2,9,8227 -2,64,2,10,8469 -2,256,1,1,35927 -2,256,1,2,37936 -2,256,1,3,36306 -2,256,1,4,36228 -2,256,1,5,36804 -2,256,1,6,35403 -2,256,1,7,35714 -2,256,1,8,36027 -2,256,1,9,36621 -2,256,1,10,36597 -2,256,2,1,31978 -2,256,2,2,31533 -2,256,2,3,31932 -2,256,2,4,31528 -2,256,2,5,31487 -2,256,2,6,31930 -2,256,2,7,33285 -2,256,2,8,31466 -2,256,2,9,32205 -2,256,2,10,31718 -2,512,1,1,66914 -2,512,1,2,68774 -2,512,1,3,68994 -2,512,1,4,67744 -2,512,1,5,67523 -2,512,1,6,69898 -2,512,1,7,69765 -2,512,1,8,69054 -2,512,1,9,68442 -2,512,1,10,68989 -2,512,2,1,60857 -2,512,2,2,60558 -2,512,2,3,62415 -2,512,2,4,61730 -2,512,2,5,62048 -2,512,2,6,61636 -2,512,2,7,61733 -2,512,2,8,61449 -2,512,2,9,61961 -2,512,2,10,62689 -3,64,1,1,9760 -3,64,1,2,9844 -3,64,1,3,9933 -3,64,1,4,9870 -3,64,1,5,9545 -3,64,1,6,9697 -3,64,1,7,9589 -3,64,1,8,9887 -3,64,1,9,9801 -3,64,1,10,9967 -3,64,2,1,8410 -3,64,2,2,8477 -3,64,2,3,8650 -3,64,2,4,11078 -3,64,2,5,16520 -3,64,2,6,12802 -3,64,2,7,12738 -3,64,2,8,8634 -3,64,2,9,8516 -3,64,2,10,8259 -3,256,1,1,36533 -3,256,1,2,36790 -3,256,1,3,36643 -3,256,1,4,35873 -3,256,1,5,38027 -3,256,1,6,36220 -3,256,1,7,36321 -3,256,1,8,36267 -3,256,1,9,36183 -3,256,1,10,35812 -3,256,2,1,31827 -3,256,2,2,32730 -3,256,2,3,32601 -3,256,2,4,33006 -3,256,2,5,32154 -3,256,2,6,32364 -3,256,2,7,33601 -3,256,2,8,32121 -3,256,2,9,32159 -3,256,2,10,32311 -3,512,1,1,70976 -3,512,1,2,68575 -3,512,1,3,70807 -3,512,1,4,69899 -3,512,1,5,82374 -3,512,1,6,67953 -3,512,1,7,70938 -3,512,1,8,69135 -3,512,1,9,69190 -3,512,1,10,69697 -3,512,2,1,61740 -3,512,2,2,62476 -3,512,2,3,60883 -3,512,2,4,60562 -3,512,2,5,62436 -3,512,2,6,61313 -3,512,2,7,60699 -3,512,2,8,63034 -3,512,2,9,62417 -3,512,2,10,60530 +trial,size_mib,rails,iteration,read_us +1,64,1,1,8827 +1,64,1,2,9182 +1,64,1,3,8795 +1,64,1,4,9193 +1,64,1,5,8816 +1,64,1,6,8891 +1,64,1,7,9217 +1,64,1,8,9066 +1,64,1,9,8983 +1,64,1,10,9134 +1,64,2,1,7881 +1,64,2,2,7936 +1,64,2,3,7749 +1,64,2,4,7721 +1,64,2,5,7656 +1,64,2,6,7804 +1,64,2,7,7749 +1,64,2,8,7712 +1,64,2,9,7758 +1,64,2,10,7709 +1,256,1,1,45615 +1,256,1,2,40748 +1,256,1,3,41477 +1,256,1,4,38106 +1,256,1,5,42597 +1,256,1,6,35767 +1,256,1,7,36386 +1,256,1,8,37482 +1,256,1,9,37504 +1,256,1,10,39268 +1,256,2,1,36651 +1,256,2,2,39218 +1,256,2,3,42937 +1,256,2,4,46352 +1,256,2,5,41977 +1,256,2,6,32533 +1,256,2,7,31392 +1,256,2,8,32557 +1,256,2,9,33837 +1,256,2,10,30433 +1,512,1,1,69993 +1,512,1,2,70742 +1,512,1,3,74133 +1,512,1,4,69588 +1,512,1,5,71132 +1,512,1,6,68560 +1,512,1,7,80041 +1,512,1,8,71435 +1,512,1,9,69264 +1,512,1,10,66838 +1,512,2,1,59802 +1,512,2,2,59422 +1,512,2,3,58280 +1,512,2,4,62639 +1,512,2,5,85828 +1,512,2,6,59767 +1,512,2,7,59856 +1,512,2,8,59810 +1,512,2,9,58579 +1,512,2,10,59135 +2,64,1,1,9204 +2,64,1,2,9158 +2,64,1,3,9223 +2,64,1,4,8809 +2,64,1,5,8824 +2,64,1,6,8740 +2,64,1,7,8794 +2,64,1,8,10210 +2,64,1,9,8845 +2,64,1,10,8597 +2,64,2,1,7693 +2,64,2,2,7713 +2,64,2,3,7536 +2,64,2,4,7630 +2,64,2,5,7726 +2,64,2,6,7572 +2,64,2,7,9093 +2,64,2,8,7682 +2,64,2,9,7640 +2,64,2,10,7632 +2,256,1,1,33653 +2,256,1,2,34287 +2,256,1,3,33790 +2,256,1,4,33883 +2,256,1,5,34287 +2,256,1,6,34441 +2,256,1,7,38144 +2,256,1,8,34870 +2,256,1,9,35909 +2,256,1,10,34287 +2,256,2,1,30481 +2,256,2,2,30234 +2,256,2,3,30088 +2,256,2,4,30409 +2,256,2,5,29880 +2,256,2,6,29918 +2,256,2,7,31259 +2,256,2,8,30141 +2,256,2,9,30128 +2,256,2,10,31197 +2,512,1,1,66601 +2,512,1,2,67722 +2,512,1,3,65813 +2,512,1,4,65718 +2,512,1,5,67617 +2,512,1,6,65629 +2,512,1,7,65590 +2,512,1,8,65847 +2,512,1,9,72218 +2,512,1,10,66336 +2,512,2,1,61289 +2,512,2,2,60220 +2,512,2,3,59870 +2,512,2,4,60168 +2,512,2,5,110346 +2,512,2,6,63599 +2,512,2,7,63203 +2,512,2,8,63166 +2,512,2,9,63773 +2,512,2,10,63886 +3,64,1,1,9491 +3,64,1,2,9538 +3,64,1,3,9957 +3,64,1,4,9986 +3,64,1,5,10197 +3,64,1,6,13359 +3,64,1,7,11310 +3,64,1,8,13719 +3,64,1,9,11336 +3,64,1,10,12525 +3,64,2,1,10317 +3,64,2,2,10817 +3,64,2,3,9657 +3,64,2,4,10324 +3,64,2,5,13524 +3,64,2,6,14548 +3,64,2,7,10219 +3,64,2,8,9108 +3,64,2,9,9123 +3,64,2,10,9123 +3,256,1,1,37821 +3,256,1,2,37920 +3,256,1,3,38634 +3,256,1,4,39309 +3,256,1,5,36401 +3,256,1,6,34913 +3,256,1,7,35350 +3,256,1,8,34702 +3,256,1,9,35080 +3,256,1,10,35031 +3,256,2,1,31144 +3,256,2,2,30864 +3,256,2,3,30997 +3,256,2,4,30970 +3,256,2,5,30867 +3,256,2,6,31096 +3,256,2,7,30902 +3,256,2,8,30845 +3,256,2,9,32227 +3,256,2,10,31013 +3,512,1,1,66056 +3,512,1,2,70733 +3,512,1,3,67038 +3,512,1,4,66280 +3,512,1,5,67983 +3,512,1,6,68702 +3,512,1,7,68377 +3,512,1,8,67651 +3,512,1,9,67569 +3,512,1,10,68441 +3,512,2,1,59335 +3,512,2,2,59728 +3,512,2,3,59263 +3,512,2,4,58559 +3,512,2,5,58660 +3,512,2,6,59220 +3,512,2,7,58917 +3,512,2,8,59252 +3,512,2,9,62268 +3,512,2,10,58719 diff --git a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv index 85860b5..f018ea8 100644 --- a/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv +++ b/kv-service/benchmarks/results/2026-10-02-independent-rail-mock-summary.csv @@ -1,19 +1,19 @@ -trial,size_mib,rails,iterations,mean_read_us,effective_gib_s,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes -1,64,1,10,10100,6.188,40708,56027,264504,671088640,0 -1,64,2,10,8276,7.552,39121,53618,264336,335544320,335544320 -1,256,1,10,36525,6.845,179122,180736,1050868,2684354560,0 -1,256,2,10,31979,7.818,185550,215482,1050864,1342177280,1342177280 -1,512,1,10,70596,7.083,353435,347682,2099584,5368709120,0 -1,512,2,10,61969,8.069,385636,407558,2099568,2684354560,2684354560 -2,64,1,10,9330,6.699,38601,50319,264504,671088640,0 -2,64,2,10,8451,7.396,46332,51551,264328,335544320,335544320 -2,256,1,10,36356,6.876,183907,175276,1050988,2684354560,0 -2,256,2,10,31906,7.836,185868,213359,1050856,1342177280,1342177280 -2,512,1,10,68609,7.288,360518,320170,2099600,5368709120,0 -2,512,2,10,61707,8.103,391285,394308,2099472,2684354560,2684354560 -3,64,1,10,9789,6.385,36027,56529,264504,671088640,0 -3,64,2,10,10408,6.005,51706,71837,264324,335544320,335544320 -3,256,1,10,36466,6.856,164191,196881,1050848,2684354560,0 -3,256,2,10,32487,7.695,189599,213688,1050876,1342177280,1342177280 -3,512,1,10,70954,7.047,376364,328773,2099592,5368709120,0 -3,512,2,10,61609,8.116,395817,387664,2099552,2684354560,2684354560 +trial,size_mib,rails,iterations,mean_read_us,effective_gib_s,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes +1,64,1,10,9010,6.937,39425,46506,264488,671088640,0 +1,64,2,10,7767,8.047,46058,43192,264344,335544320,335544320 +1,256,1,10,39495,6.33,208973,181504,1050960,2684354560,0 +1,256,2,10,36788,6.796,225725,234681,1050924,1342177280,1342177280 +1,512,1,10,71172,7.025,386817,320400,2099600,5368709120,0 +1,512,2,10,62311,8.024,399810,408032,2099456,2684354560,2684354560 +2,64,1,10,9040,6.914,46418,38929,264536,671088640,0 +2,64,2,10,7791,8.022,42413,50651,264312,335544320,335544320 +2,256,1,10,34755,7.193,183867,158279,1050880,2684354560,0 +2,256,2,10,30373,8.231,181888,199803,1050856,1342177280,1342177280 +2,512,1,10,66909,7.473,338050,325553,2099588,5368709120,0 +2,512,2,10,66952,7.468,402060,482669,2099332,2684354560,2684354560 +3,64,1,10,11141,5.61,49556,54652,264492,671088640,0 +3,64,2,10,10676,5.854,55339,69350,264424,335544320,335544320 +3,256,1,10,36516,6.846,194192,165791,1050940,2684354560,0 +3,256,2,10,31092,8.041,179116,210696,1051004,1342177280,1342177280 +3,512,1,10,67883,7.366,350056,325666,2099604,5368709120,0 +3,512,2,10,59392,8.419,397439,351243,2099532,2684354560,2684354560 From 1bd482101c0013ae82679ac2c11ced9547cce543 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 12:02:36 +0800 Subject: [PATCH 07/20] Validate descriptor reads over a real HCA and harden completion handling Implement Tier A pointer streaming for slab-backed striped reads, reject missing completion bytes, and retain uncertain slab allocations through QP teardown. Count failed CQEs when harvesting the queue to avoid an unnecessary final poll timeout. Add isolated fault-injection hooks and hardware-gated tests for single-rail success, disconnect, cancellation, checksum corruption, and late WRITEs, plus a validation config. --- README.md | 21 ++- kv-service/client-rs/tests/rail_read_e2e.rs | 173 +++++++++++++++++- .../configs/server-rail-validation.toml | 52 ++++++ kv-service/server/src/io_executor/tier_a.rs | 149 +++++++++++++++ kv-service/server/src/rdma/qp.rs | 53 ++++-- kv-service/server/src/rdma/server.rs | 57 +++++- 6 files changed, 481 insertions(+), 24 deletions(-) create mode 100644 kv-service/configs/server-rail-validation.toml diff --git a/README.md b/README.md index 88b0c01..a52d666 100644 --- a/README.md +++ b/README.md @@ -340,11 +340,10 @@ validation, enable `verify_stripe_checksums` on the server and rewrite the objects being tested so their placements contain checksums. With the server's default setting, older placements may have no checksums. Cancellation and publishing share a gate, so whichever starts first determines the result; -a failed or cancelled read -leaves its caller buffer unchanged. Each rail owns a compact registered -receive buffer. The reader bounds concurrent tasks and aggregate in-flight -bytes per rail across simultaneous requests. It retains the active-read and -final-staging budget through the post-read lookup and publish step; transfer +a failed or cancelled read leaves its caller buffer unchanged. Each rail owns +a compact registered receive buffer. The reader bounds concurrent tasks and +aggregate in-flight bytes per rail across simultaneous requests. It retains +the active-read and final-staging budget through the post-read lookup and publish step; transfer and MR reservations end after all rail workers finish. No in-request transparent retry is attempted. The server's stripe-subset fallback also honors tag-15 scatter destinations @@ -353,8 +352,16 @@ uncertain completion retains its source and MR until its QP is destroyed. For hardware-independent scheduling and failure checks, run `cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma -rail_read::tests`. The ignored `rail_read_e2e` tests require two reachable -RDMA listeners and the `CS_RAIL_*` endpoint/device environment variables. +rail_read::tests`. The ignored `rail_read_e2e` suite contains five single +real-Rail checks and three dual-Rail checks. Select a test with `--ignored +--exact --nocapture` and configure `CS_RAIL_COORDINATOR`, +`CS_RAIL_LISTENER0/1`, `CS_RAIL_DEVICE0/1`, and `CS_RAIL_GID0/1` as needed. +`CS_RDMA_SLAB_MB=0` on the server exercises the registered-buffer fallback. +On an isolated server only, `CS_RDMA_TEST_PRE_WRITE_DELAY_MS=3000` delays +stripe-subset WRITEs for the late-completion test; the delay is bounded to +five seconds and must be unset after that test. A physical-stripe corruption +test additionally requires checksum verification enabled before writing its +object and a reversible fault injection into one test-only stripe file. The ignored `software_only_mock_benchmark` exercises scheduling and memory copies; its throughput is **not** an RDMA hardware result. Reproduce its paired 1/2-rail matrix and save all per-read samples with diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs index b456c79..246fbb8 100644 --- a/kv-service/client-rs/tests/rail_read_e2e.rs +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -5,12 +5,12 @@ #![cfg(feature = "rdma")] -use contextstore_client_rs::rail_read::{RailLimits, RailReader, RailRoute}; +use contextstore_client_rs::rail_read::{RailCancel, RailLimits, RailReader, RailRoute}; use contextstore_client_rs::rdma::RdmaClientConfig; use contextstore_client_rs::KvClient; use prost::bytes::Bytes; use std::sync::Arc; -use std::time::{Duration, SystemTime, UNIX_EPOCH}; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; fn setting(name: &str, default: &str) -> String { std::env::var(name).unwrap_or_else(|_| default.to_string()) @@ -83,6 +83,175 @@ async fn seeded_object() -> (KvClient, String, Vec, String) { (client, key, payload, advertised) } +#[tokio::test] +#[ignore = "requires one reachable HCA or RXE listener and a striped KVService"] +async fn single_real_rail_restores_verified_object() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener = setting("CS_RAIL_LISTENER0", &advertised); + let reader = Arc::new( + RailReader::new( + vec![route(&advertised, 0, &listener)], + RailLimits::default(), + ) + .expect("single rail"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert_eq!( + client + .read_multi_rail_into( + Arc::clone(&reader), + "rail-e2e", + &key, + &mut destination, + None + ) + .await + .expect("real rail read"), + Some(payload.len()) + ); + assert_eq!(destination, payload); + assert_eq!(reader.snapshots()[0].bytes, payload.len() as u64); +} + +#[tokio::test] +#[ignore = "requires a striped KVService and an injected dead RDMA listener"] +async fn dead_single_real_rail_leaves_destination_unchanged() { + let (mut client, key, payload, advertised) = seeded_object().await; + let dead = setting("CS_RAIL_DEAD_LISTENER", "127.0.0.1:59999"); + let reader = Arc::new( + RailReader::new( + vec![route(&advertised, 0, &dead)], + RailLimits { + io_timeout: Duration::from_secs(3), + ..RailLimits::default() + }, + ) + .expect("dead rail"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert!(client + .read_multi_rail_into(reader, "rail-e2e", &key, &mut destination, None) + .await + .is_err()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} + +#[tokio::test] +#[ignore = "requires one reachable HCA or RXE listener and a large striped object"] +async fn cancellation_after_real_transfer_starts_cannot_write_reused_buffer() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener = setting("CS_RAIL_LISTENER0", &advertised); + let reader = Arc::new( + RailReader::new( + vec![route(&advertised, 0, &listener)], + RailLimits::default(), + ) + .expect("single rail"), + ); + let cancel = RailCancel::default(); + let task_reader = Arc::clone(&reader); + let task_cancel = cancel.clone(); + let read_task = tokio::spawn(async move { + let mut destination = vec![0xA5; payload.len()]; + let result = client + .read_multi_rail_into( + task_reader, + "rail-e2e", + &key, + &mut destination, + Some(task_cancel), + ) + .await; + (result, destination) + }); + let deadline = Instant::now() + Duration::from_secs(10); + while reader.snapshots()[0].inflight_requests == 0 { + assert!( + Instant::now() < deadline, + "real RDMA transfer never started" + ); + tokio::time::sleep(Duration::from_millis(1)).await; + } + cancel.cancel(); + let (result, mut destination) = tokio::time::timeout(Duration::from_secs(40), read_task) + .await + .expect("cancelled transfer did not quiesce") + .expect("read task joined"); + assert!(result + .expect_err("cancelled transfer must fail") + .to_string() + .contains("cancelled")); + assert!(destination.iter().all(|byte| *byte == 0xA5)); + destination.fill(0x33); + tokio::time::sleep(Duration::from_millis(100)).await; + assert!(destination.iter().all(|byte| *byte == 0x33)); + assert_eq!(reader.snapshots()[0].inflight_requests, 0); + assert_eq!(reader.snapshots()[0].registered_bytes, 0); +} + +#[tokio::test] +#[ignore = "requires an existing striped object with one deliberately corrupted physical stripe"] +async fn corrupted_real_stripe_does_not_publish_bytes() { + let coordinator = setting("CS_RAIL_COORDINATOR", "http://127.0.0.1:50051"); + let namespace = setting("CS_RAIL_EXISTING_NAMESPACE", "rust-bench"); + let key = setting("CS_RAIL_EXISTING_KEY", "railtest0/__combined__"); + let mut client = KvClient::connect(coordinator).await.expect("connect gRPC"); + let lookup = client + .lookup_object(&namespace, &key) + .await + .expect("lookup") + .expect("object exists"); + let placement = lookup.placement.expect("placement"); + assert!( + placement + .chunks + .iter() + .all(|chunk| !chunk.checksum.is_empty()), + "checksum injection requires a newly written, checksummed object" + ); + let advertised = placement.chunks[0].rdma_endpoint.clone(); + let listener = setting("CS_RAIL_LISTENER0", &advertised); + let reader = Arc::new( + RailReader::new( + vec![route(&advertised, 0, &listener)], + RailLimits::default(), + ) + .expect("single rail"), + ); + let mut destination = vec![0xA5; lookup.descriptor.size as usize]; + assert!(client + .read_multi_rail_into(reader, &namespace, &key, &mut destination, None) + .await + .is_err()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} + +#[tokio::test] +#[ignore = "requires CS_RDMA_TEST_PRE_WRITE_DELAY_MS=3000 on an isolated real RDMA server"] +async fn late_server_write_after_timeout_cannot_corrupt_reused_buffer() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener = setting("CS_RAIL_LISTENER0", &advertised); + let reader = Arc::new( + RailReader::new( + vec![route(&advertised, 0, &listener)], + RailLimits { + io_timeout: Duration::from_secs(1), + ..RailLimits::default() + }, + ) + .expect("single rail"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert!(client + .read_multi_rail_into(reader, "rail-e2e", &key, &mut destination, None) + .await + .is_err()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); + destination.fill(0x33); + tokio::time::sleep(Duration::from_secs(4)).await; + assert!(destination.iter().all(|byte| *byte == 0x33)); +} + #[tokio::test] #[ignore = "requires two reachable RDMA listeners on the same storage node"] async fn same_object_matches_single_and_dual_rail() { diff --git a/kv-service/configs/server-rail-validation.toml b/kv-service/configs/server-rail-validation.toml new file mode 100644 index 0000000..510bd86 --- /dev/null +++ b/kv-service/configs/server-rail-validation.toml @@ -0,0 +1,52 @@ +# Isolated rail validation. Set reachable API/RDMA addresses, storage paths, +# and the dedicated Redis port before use. Configure listeners separately via +# CS_RDMA_DEVICES=device:ip:port:gid[,device:ip:port:gid]. +[api] +listen = "127.0.0.1:55151" +max_connections = 100 + +[cluster] +node_id = "rail-validation" +grpc_advertise = "127.0.0.1:55151" +rdma_advertise = "127.0.0.1:55153" +data_nodes = [] + +[storage] +# Distinct directories exercise striping logic; use separate NVMe mounts to +# measure disk aggregation rather than one filesystem's cache. +devices = ["./data/rail0", "./data/rail1"] +data_subdir = "contextstore" +striping_threshold = 4194304 +striping_chunk_size = 4194304 +rdma_stream_chunk_size = 1048576 +verify_stripe_checksums = true + +[memory_tier] +capacity_mb = 128 +slab_size_mb = 4 +use_pinned_memory = false + +[io_executor] +kind = "tier_a" +thread_pool_size = 8 +io_uring_depth = 32 + +[router] +strategy = "object_hash" + +[metadata] +redis_url = "redis://127.0.0.1:6388/" +redis_key_prefix = "contextstore:rail-validation:" +redis_connect_timeout_ms = 1000 +redis_command_timeout_ms = 1000 + +[gc] +enabled = false +interval_seconds = 300 +grace_seconds = 600 +max_tasks_per_run = 1000 +task_lease_seconds = 300 + +[metrics] +enabled = false +listen = "127.0.0.1:55190" diff --git a/kv-service/server/src/io_executor/tier_a.rs b/kv-service/server/src/io_executor/tier_a.rs index 7c8816c..4ef62de 100644 --- a/kv-service/server/src/io_executor/tier_a.rs +++ b/kv-service/server/src/io_executor/tier_a.rs @@ -60,6 +60,15 @@ enum Job { path: std::path::PathBuf, resp: channel::Sender>, }, + /// Read a physical range directly into caller-owned staging memory. + /// Completion must be consumed before the caller can release the pointer. + ReadIntoPtr { + req: IORequest, + ptr: MutPtrWrapper, + capacity: usize, + index: usize, + resp: channel::Sender<(usize, Result)>, + }, Shutdown, } @@ -69,6 +78,9 @@ enum Job { struct PtrWrapper(*const u8); unsafe impl Send for PtrWrapper {} +struct MutPtrWrapper(*mut u8); +unsafe impl Send for MutPtrWrapper {} + impl TierAExecutor { pub fn new(num_workers: usize) -> Self { let (tx, rx) = channel::unbounded::(); @@ -124,6 +136,30 @@ impl TierAExecutor { Ok(Job::ReadAligned { path, resp }) => { let _ = resp.send(read_aligned_impl(&path)); } + Ok(Job::ReadIntoPtr { + req, + ptr, + capacity, + index, + resp, + }) => { + let result = read_into_ptr_impl(&req, ptr.0, capacity); + if let Err(error) = &result { + log_io_error( + IoLogContext { + executor: "tier_a", + operation: "read", + mode: "into_ptr_stream", + device_id: -1, + job_id: index as u64, + }, + &req, + req.length, + error, + ); + } + let _ = resp.send((index, result)); + } Ok(Job::Shutdown) | Err(_) => break, } })); @@ -168,6 +204,33 @@ fn read_file_impl(req: &IORequest) -> Result> { Ok(buf) } +fn read_into_ptr_impl(req: &IORequest, ptr: *mut u8, capacity: usize) -> Result { + use std::io::{Read, Seek, SeekFrom}; + + let mut file = std::fs::File::open(&req.path)?; + let length = if req.length == 0 { + let file_size = file.metadata()?.len(); + usize::try_from(file_size.saturating_sub(req.offset)) + .map_err(|_| KVError::InvalidArgument("file range exceeds address space".into()))? + } else { + req.length + }; + if length > capacity { + return Err(KVError::InvalidArgument(format!( + "pointer read needs {length} bytes but capacity is {capacity}" + ))); + } + if length == 0 { + return Ok(0); + } + file.seek(SeekFrom::Start(req.offset))?; + // SAFETY: IOExecutor's pointer contract requires writable memory through + // completion. This worker sends its completion only after read_exact ends. + let target = unsafe { std::slice::from_raw_parts_mut(ptr, length) }; + file.read_exact(target)?; + Ok(length) +} + fn write_file_impl(req: &IORequest, data: &[u8]) -> Result<()> { use std::io::Write; if let Some(parent) = req.path.parent() { @@ -822,6 +885,48 @@ impl IOExecutor for TierAExecutor { ); results } + + fn read_aligned_into_ptr_batch( + &self, + requests: Vec<(IORequest, *mut u8, usize)>, + ) -> Vec> { + let expected = requests.len(); + let mut completed: Vec>> = + std::iter::repeat_with(|| None).take(expected).collect(); + for (index, result) in self.read_aligned_into_ptr_stream(requests) { + if let Some(slot) = completed.get_mut(index) { + *slot = Some(result); + } + } + completed + .into_iter() + .map(|result| { + result.unwrap_or_else(|| { + Err(KVError::Internal("pointer read completion missing".into())) + }) + }) + .collect() + } + + fn read_aligned_into_ptr_stream( + &self, + requests: Vec<(IORequest, *mut u8, usize)>, + ) -> channel::Receiver<(usize, Result)> { + let (sender, receiver) = channel::unbounded(); + for (index, (req, ptr, capacity)) in requests.into_iter().enumerate() { + let request = Job::ReadIntoPtr { + req, + ptr: MutPtrWrapper(ptr), + capacity, + index, + resp: sender.clone(), + }; + if self.sender.send(request).is_err() { + let _ = sender.send((index, Err(KVError::Internal("executor closed".into())))); + } + } + receiver + } } #[cfg(test)] @@ -866,4 +971,48 @@ mod tests { assert_eq!(r.as_ref().unwrap(), format!("data-{}", i).as_bytes()); } } + + #[test] + fn pointer_stream_reports_every_range_and_fills_aligned_destination() { + let tmp = TempDir::new().unwrap(); + let exec = TierAExecutor::new(2); + let first = tmp.path().join("first.bin"); + let second = tmp.path().join("second.bin"); + exec.write_file(&first, &vec![0x11; DIRECT_IO_ALIGN]) + .unwrap(); + exec.write_file(&second, &vec![0x22; DIRECT_IO_ALIGN]) + .unwrap(); + let mut buffer = AlignedBuffer::new(2 * DIRECT_IO_ALIGN, DIRECT_IO_ALIGN); + let base = buffer.as_mut_ptr(); + let completions: Vec<_> = exec + .read_aligned_into_ptr_stream(vec![ + ( + IORequest { + path: first, + offset: 0, + length: DIRECT_IO_ALIGN, + }, + base, + DIRECT_IO_ALIGN, + ), + ( + IORequest { + path: second, + offset: 0, + length: DIRECT_IO_ALIGN, + }, + unsafe { base.add(DIRECT_IO_ALIGN) }, + DIRECT_IO_ALIGN, + ), + ]) + .iter() + .collect(); + assert_eq!(completions.len(), 2); + for (index, result) in completions { + assert_eq!(result.unwrap(), DIRECT_IO_ALIGN, "range {index}"); + } + let bytes = unsafe { std::slice::from_raw_parts(base, 2 * DIRECT_IO_ALIGN) }; + assert!(bytes[..DIRECT_IO_ALIGN].iter().all(|byte| *byte == 0x11)); + assert!(bytes[DIRECT_IO_ALIGN..].iter().all(|byte| *byte == 0x22)); + } } diff --git a/kv-service/server/src/rdma/qp.rs b/kv-service/server/src/rdma/qp.rs index 69af176..4a19376 100644 --- a/kv-service/server/src/rdma/qp.rs +++ b/kv-service/server/src/rdma/qp.rs @@ -1,6 +1,7 @@ //! Reliable Connection QP — one-to-one connection with a client; use ibv_post_send WRITE after handshake use crate::rdma::context::{MemRegion, RdmaContext}; +use crate::rdma::slab::SlabExtent; use anyhow::{anyhow, Result}; use prost::bytes::Bytes; use rdma_sys::*; @@ -20,6 +21,7 @@ pub struct RcQp { // A timed-out WRITE may still read its local MR. These buffers are // released only after Drop destroys the QP. retired_writes: Mutex>, + retired_extents: Mutex>, /// Local QP info, sent to remote over the control plane pub local: QpInfo, } @@ -113,6 +115,7 @@ impl RcQp { Ok(Self { qp, retired_writes: Mutex::new(Vec::new()), + retired_extents: Mutex::new(Vec::new()), local, }) } @@ -295,7 +298,7 @@ impl RcQp { /// 非阻塞收割: poll 当前已完成的 CQE (最多 `max`), 立即返回收到的数量. /// 用于流水线中的机会式回收 — post 间隙顺手清 CQ, 避免凑满窗口后长阻塞. - pub fn poll_available(cq: NonNull, max: usize) -> Result { + pub fn poll_available(cq: NonNull, max: usize) -> Result<(usize, Option)> { unsafe { let cap = max.max(1); let mut wcs: Vec = Vec::with_capacity(cap); @@ -306,25 +309,36 @@ impl RcQp { if n < 0 { return Err(anyhow!("ibv_poll_cq error")); } - for wc in wcs.iter().take(n as usize) { - if wc.status != ibv_wc_status::IBV_WC_SUCCESS { - return Err(anyhow!( - "WR {} failed: status={} ({})", - wc.wr_id, - wc.status, - wc_status_str(wc.status), - )); - } - } - Ok(n as usize) + Ok(Self::summarize_completions(&wcs[..n as usize])) } } + fn summarize_completions(wcs: &[ibv_wc]) -> (usize, Option) { + let first_error = wcs + .iter() + .find(|wc| wc.status != ibv_wc_status::IBV_WC_SUCCESS) + .map(|wc| { + format!( + "WR {} failed: status={} ({})", + wc.wr_id, + wc.status, + wc_status_str(wc.status) + ) + }); + (wcs.len(), first_error) + } + /// Retain an uncertain local WRITE source until this QP is destroyed. /// Used when polling cannot prove a signaled WRITE has quiesced. pub fn retain_uncertain_write(&self, mr: MemRegion, source: Bytes) { self.retired_writes.lock().unwrap().push((mr, source)); } + + /// Keep a slab allocation pinned until QP destruction when a CQ poll + /// cannot prove every WRITE has stopped using it. + pub fn retain_uncertain_extent(&self, extent: SlabExtent) { + self.retired_extents.lock().unwrap().push(extent); + } } fn wc_status_str(status: u32) -> &'static str { @@ -349,3 +363,18 @@ impl Drop for RcQp { } } } + +#[cfg(test)] +mod completion_tests { + use super::*; + + #[test] + fn failed_cqe_is_still_counted_as_consumed() { + let mut completion: ibv_wc = unsafe { std::mem::zeroed() }; + completion.wr_id = 9; + completion.status = 12; + let (count, error) = RcQp::summarize_completions(&[completion]); + assert_eq!(count, 1); + assert!(error.expect("failed status").contains("WR 9 failed")); + } +} diff --git a/kv-service/server/src/rdma/server.rs b/kv-service/server/src/rdma/server.rs index cfce624..4291bf2 100644 --- a/kv-service/server/src/rdma/server.rs +++ b/kv-service/server/src/rdma/server.rs @@ -1110,6 +1110,15 @@ fn fallback_write_targets( } } +fn ensure_subset_complete(expected_bytes: usize, actual_bytes: u64) -> Result<()> { + if actual_bytes != expected_bytes as u64 { + return Err(anyhow!( + "stripe-subset stream incomplete: expected {expected_bytes} bytes, wrote {actual_bytes}" + )); + } + Ok(()) +} + fn serve_get_stripes_fallback( kv_ctx: &Arc, rdma: &Arc, @@ -1298,6 +1307,21 @@ fn serve_get_stripes( } }; let stream_setup_us = stream_start.elapsed().as_micros() as u64; + // Isolated hardware fault injection: let the client time out and retire + // its QP/MR before this server attempts a late WRITE. Unset in production. + if let Ok(raw_delay) = std::env::var("CS_RDMA_TEST_PRE_WRITE_DELAY_MS") { + if let Ok(delay_ms) = raw_delay.parse::() { + if delay_ms > 0 { + let bounded_ms = delay_ms.min(5_000); + tracing::warn!( + event = "rdma_test_pre_write_delay", + delay_ms = bounded_ms, + "delaying stripe-subset WRITEs for fault injection" + ); + std::thread::sleep(std::time::Duration::from_millis(bounded_ms)); + } + } + } let view = extent.view(nic_idx); const COMPLETION_WINDOW: usize = RcQp::MAX_SEND_WR / 2; let mut outstanding = 0usize; @@ -1307,6 +1331,7 @@ fn serve_get_stripes( let mut first_io_us = None; let mut last_io_us = 0u64; let mut first_error = None; + let mut uncertain_write = false; while let Ok((stripe_index, source_offset, object_offset, length, result)) = stream.recv() { let completion_us = stream_start.elapsed().as_micros() as u64; @@ -1362,8 +1387,16 @@ fn serve_get_stripes( // 完成事件 → 流水线断流. 现在仅当 SQ 接近满时才阻塞等 1 个腾位. let poll_start = std::time::Instant::now(); match RcQp::poll_available(client_cq, outstanding) { - Ok(n) => outstanding -= n, + Ok((n, completion_error)) => { + outstanding -= n; + if let Some(error) = completion_error { + uncertain_write = true; + first_error = + Some(format!("poll RDMA completion window: {error}")); + } + } Err(error) => { + uncertain_write = true; first_error = Some(format!("poll RDMA completion window: {error}")) } @@ -1372,6 +1405,7 @@ fn serve_get_stripes( match RcQp::poll_n(client_cq, 1) { Ok(()) => outstanding -= 1, Err(error) => { + uncertain_write = true; first_error = Some(format!("poll RDMA completion window: {error}")) } @@ -1392,14 +1426,20 @@ fn serve_get_stripes( } } - if outstanding > 0 { + if outstanding > 0 && !uncertain_write { let poll_start = std::time::Instant::now(); let poll_result = RcQp::poll_n(client_cq, outstanding); poll_us += poll_start.elapsed().as_micros() as u64; if let Err(error) = poll_result { + uncertain_write = true; first_error.get_or_insert_with(|| format!("poll final RDMA completions: {error}")); } } + if first_error.is_none() { + if let Err(error) = ensure_subset_complete(staged_bytes, total) { + first_error = Some(error.to_string()); + } + } let total_us = total_start.elapsed().as_micros() as u64; tracing::info!( "RDMA_SUBSET_DETAIL key={} bytes={} stripes={} requests={} alloc_us={} stream_setup_us={} first_io_us={} last_io_us={} poll_us={} total_us={}", @@ -1416,6 +1456,9 @@ fn serve_get_stripes( ); if let Some(error) = first_error { + if uncertain_write { + qp.retain_uncertain_extent(extent); + } tracing::warn!( key = %kv_key.to_string_key(), error = %error, @@ -2062,7 +2105,7 @@ mod tests { #[cfg(test)] mod sge_tests { - use super::{fallback_write_targets, map_range_to_segments}; + use super::{ensure_subset_complete, fallback_write_targets, map_range_to_segments}; #[test] fn range_within_one_segment() { @@ -2108,4 +2151,12 @@ mod sge_tests { vec![(0x9000, 12, 64)] ); } + + #[test] + fn missing_or_duplicate_stream_completions_cannot_report_success() { + assert!(ensure_subset_complete(64, 0).is_err()); + assert!(ensure_subset_complete(64, 32).is_err()); + assert!(ensure_subset_complete(64, 128).is_err()); + assert!(ensure_subset_complete(64, 64).is_ok()); + } } From 8245ae45a50e4c28bd7dad500030a0c83720808c Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 12:08:38 +0800 Subject: [PATCH 08/20] Publish physical single-rail RDMA measurements and topology limits Record exact 64 MiB sample latencies for slab and fallback modes, HCA port-counter snapshots, fault-injection outcomes, and the shared-SATA storage boundary. Keep the hardware data separate from the software-only two-rail Mock matrix. --- README.md | 3 + .../results/2026-10-02-skv-single-hca.json | 115 ++++++++++++++++++ 2 files changed, 118 insertions(+) create mode 100644 kv-service/benchmarks/results/2026-10-02-skv-single-hca.json diff --git a/README.md b/README.md index a52d666..b73ba0a 100644 --- a/README.md +++ b/README.md @@ -368,6 +368,9 @@ paired 1/2-rail matrix and save all per-read samples with `python kv-service/benchmarks/collect_rail_mock.py --sizes 64,256,512`. The checked-in `kv-service/benchmarks/results/2026-10-02-independent-rail-mock-*` files contain one Linux ARM64 software-only run, including environment details. +The separate `kv-service/benchmarks/results/2026-10-02-skv-single-hca.json` +records physical single-rail Verbs reads, HCA port-counter deltas, fault tests, +and the storage/topology boundary. It does not measure two-rail aggregation. --- diff --git a/kv-service/benchmarks/results/2026-10-02-skv-single-hca.json b/kv-service/benchmarks/results/2026-10-02-skv-single-hca.json new file mode 100644 index 0000000..f84cde5 --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-skv-single-hca.json @@ -0,0 +1,115 @@ +{ + "schema_version": 1, + "label": "physical RDMA, one active HCA rail; no multi-rail hardware scaling claim", + "source_commit": "1bd4821", + "environment": { + "server": "skv-node1", + "client": "skv-node2", + "hca": "Mellanox ConnectX-6 Dx mlx5_1, port 1, RoCE v2 GID index 3", + "link_speed_mbps": 100000, + "numa_node": 1, + "pci_bdf": "0000:af:00.1", + "cpu": "2 sockets, 80 logical CPUs, Intel Xeon Gold 5218R", + "server_storage": "two stripe directories on one /home ext4 filesystem backed by /dev/sde Intel SSDSC2BB96 SATA SSD", + "io_executor": "tier_a", + "redis": "dedicated local test instance", + "verify_stripe_checksums": true, + "object_bytes": 67108864, + "stripe_bytes": 4194304, + "stripes": 16, + "object_xxh3": "a0a4cbfa5cad46af" + }, + "single_rail_runs": [ + { + "mode": "slab_128_mib", + "warmup": 1, + "iterations": 5, + "latency_us": [ + 164867, + 163271, + 159849, + 164179, + 142757 + ], + "median_us": 163271, + "client_cpu_user_us_total": 288946, + "client_cpu_system_us_total": 313377, + "client_peak_rss_kb": 203896, + "rail_payload_bytes": 335544320 + }, + { + "mode": "registered_buffer_fallback_slab_0", + "warmup": 1, + "iterations": 5, + "latency_us": [ + 342838, + 345929, + 323655, + 324635, + 321299 + ], + "median_us": 324635, + "client_cpu_user_us_total": 316567, + "client_cpu_system_us_total": 285431, + "client_peak_rss_kb": 203856, + "rail_payload_bytes": 335544320 + } + ], + "hca_counter_probe": { + "mode": "slab_128_mib", + "warmup": 0, + "iterations": 5, + "latency_us": [ + 168286, + 157781, + 158097, + 140329, + 142884 + ], + "median_us": 157781, + "payload_bytes": 335544320, + "counter_path": "/sys/class/infiniband/mlx5_1/ports/1/counters/{port_xmit_data,port_rcv_data}", + "counter_unit_bytes": 4, + "counter_unit_reference": "https://www.kernel.org/doc/html/v5.15/admin-guide/abi-stable.html", + "server_before": [ + 53088016198567, + 125676323529852 + ], + "server_after": [ + 53088104837287, + 125676323565277 + ], + "client_before": [ + 1311584136149, + 2139962965962 + ], + "client_after": [ + 1311584169427, + 2140051932362 + ], + "server_delta_bytes": { + "transmitted": 354554880, + "received": 141700 + }, + "client_delta_bytes": { + "transmitted": 133112, + "received": 355865600 + }, + "caveat": "Port counters include any concurrent traffic and protocol overhead; these close deltas are evidence of HCA traffic, not isolated link saturation." + }, + "hardware_gated_tests": { + "single_rail_content": "passed, byte-equal 64 MiB object", + "dead_listener": "passed, caller buffer unchanged", + "cancel_after_inflight": "passed, 256 MiB transfer quiesced and reused buffer unchanged", + "checksum_corruption": "passed, stripe 0 mismatch detected; backup restored with matching SHA-256", + "late_write_after_timeout": "passed, one-second client deadline versus three-second server pre-WRITE delay; reused buffer unchanged", + "recovery_after_fault": "passed, normal read after removing delay", + "dual_independent_rails": "not run: every SKV node has only one active physical HCA port; RXE kernel module mismatched loaded OFED ib_core" + }, + "limitations": [ + "Both stripe directories share one SATA SSD; no multi-NVMe bandwidth evidence.", + "Tier A pointer reads may hit the Linux page cache; these samples are not cold-disk throughput.", + "One HCA rail only; no two-rail physical or Soft-RoCE expansion result.", + "The late-WRITE test observes buffer reuse after timeout but does not expose a CQE timestamp on the client." + ] +} From 45e912b635ad1d91dc9b2179fabcf9e9ff529ee9 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 12:19:29 +0800 Subject: [PATCH 09/20] Add paired x86 software-only rail benchmark evidence Run the same 64/256/512 MiB one- and two-rail Mock matrix on SKV x86_64, preserving 180 per-read samples, run summaries, and an environment manifest separately from physical HCA measurements. --- README.md | 3 + ...6-10-02-skv-x86-rail-mock-environment.json | 17 ++ .../2026-10-02-skv-x86-rail-mock-samples.csv | 181 ++++++++++++++++++ .../2026-10-02-skv-x86-rail-mock-summary.csv | 19 ++ 4 files changed, 220 insertions(+) create mode 100644 kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-environment.json create mode 100644 kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-samples.csv create mode 100644 kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-summary.csv diff --git a/README.md b/README.md index b73ba0a..45c9b8d 100644 --- a/README.md +++ b/README.md @@ -368,6 +368,9 @@ paired 1/2-rail matrix and save all per-read samples with `python kv-service/benchmarks/collect_rail_mock.py --sizes 64,256,512`. The checked-in `kv-service/benchmarks/results/2026-10-02-independent-rail-mock-*` files contain one Linux ARM64 software-only run, including environment details. +The `2026-10-02-skv-x86-rail-mock-*` files repeat the same paired matrix on +an x86_64 two-socket host; their much lower absolute copy rate is a reminder +that Mock results are host-specific software measurements. The separate `kv-service/benchmarks/results/2026-10-02-skv-single-hca.json` records physical single-rail Verbs reads, HCA port-counter deltas, fault tests, and the storage/topology boundary. It does not measure two-rail aggregation. diff --git a/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-environment.json b/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-environment.json new file mode 100644 index 0000000..29ef379 --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-environment.json @@ -0,0 +1,17 @@ +{ + "environment": "Mock software-only; no RDMA Verbs or disk I/O", + "environment_note": "skv-node1 x86_64 Xeon Gold 5218R; Mock copy only, no RDMA or disk", + "git_commit": "8245ae4", + "platform": "Linux-5.4.0-80-generic-x86_64-with-glibc2.29", + "machine": "x86_64", + "rustc": "rustc 1.90.0 (1159e78c4 2025-09-14)", + "sizes_mib": [ + 64, + 256, + 512 + ], + "trials": 3, + "iterations": 10, + "stripe_mib": 4, + "warmup_reads": 1 +} diff --git a/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-samples.csv b/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-samples.csv new file mode 100644 index 0000000..71b55c8 --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-samples.csv @@ -0,0 +1,181 @@ +trial,size_mib,rails,iteration,read_us +1,64,1,1,134005 +1,64,1,2,133905 +1,64,1,3,133546 +1,64,1,4,129405 +1,64,1,5,130268 +1,64,1,6,126075 +1,64,1,7,128052 +1,64,1,8,141236 +1,64,1,9,129860 +1,64,1,10,132194 +1,64,2,1,108229 +1,64,2,2,107791 +1,64,2,3,108477 +1,64,2,4,107764 +1,64,2,5,109471 +1,64,2,6,109478 +1,64,2,7,113054 +1,64,2,8,120586 +1,64,2,9,106842 +1,64,2,10,112745 +1,256,1,1,483642 +1,256,1,2,487601 +1,256,1,3,478132 +1,256,1,4,543857 +1,256,1,5,482796 +1,256,1,6,487772 +1,256,1,7,479935 +1,256,1,8,526729 +1,256,1,9,495042 +1,256,1,10,487211 +1,256,2,1,397504 +1,256,2,2,398836 +1,256,2,3,397197 +1,256,2,4,440456 +1,256,2,5,403665 +1,256,2,6,399278 +1,256,2,7,406916 +1,256,2,8,450101 +1,256,2,9,407137 +1,256,2,10,403248 +1,512,1,1,951262 +1,512,1,2,956457 +1,512,1,3,956890 +1,512,1,4,1020075 +1,512,1,5,1036109 +1,512,1,6,952697 +1,512,1,7,993516 +1,512,1,8,1000804 +1,512,1,9,954124 +1,512,1,10,948699 +1,512,2,1,767884 +1,512,2,2,770391 +1,512,2,3,786015 +1,512,2,4,1105444 +1,512,2,5,931722 +1,512,2,6,823808 +1,512,2,7,840925 +1,512,2,8,774664 +1,512,2,9,827244 +1,512,2,10,788150 +2,64,1,1,127421 +2,64,1,2,130462 +2,64,1,3,129895 +2,64,1,4,129856 +2,64,1,5,129486 +2,64,1,6,150049 +2,64,1,7,209316 +2,64,1,8,199248 +2,64,1,9,132685 +2,64,1,10,137070 +2,64,2,1,113138 +2,64,2,2,107672 +2,64,2,3,107558 +2,64,2,4,106566 +2,64,2,5,106363 +2,64,2,6,113607 +2,64,2,7,107671 +2,64,2,8,117632 +2,64,2,9,107032 +2,64,2,10,108744 +2,256,1,1,481912 +2,256,1,2,481889 +2,256,1,3,482913 +2,256,1,4,544547 +2,256,1,5,484146 +2,256,1,6,483429 +2,256,1,7,493390 +2,256,1,8,521778 +2,256,1,9,487144 +2,256,1,10,477209 +2,256,2,1,401716 +2,256,2,2,444675 +2,256,2,3,482788 +2,256,2,4,696733 +2,256,2,5,424365 +2,256,2,6,449166 +2,256,2,7,459335 +2,256,2,8,457407 +2,256,2,9,399861 +2,256,2,10,439317 +2,512,1,1,962279 +2,512,1,2,954041 +2,512,1,3,945880 +2,512,1,4,1009387 +2,512,1,5,1016059 +2,512,1,6,953889 +2,512,1,7,988657 +2,512,1,8,997310 +2,512,1,9,948267 +2,512,1,10,933344 +2,512,2,1,772625 +2,512,2,2,780020 +2,512,2,3,778984 +2,512,2,4,807234 +2,512,2,5,803489 +2,512,2,6,778407 +2,512,2,7,832380 +2,512,2,8,781570 +2,512,2,9,779126 +2,512,2,10,803776 +3,64,1,1,128889 +3,64,1,2,126413 +3,64,1,3,133012 +3,64,1,4,127886 +3,64,1,5,128800 +3,64,1,6,126971 +3,64,1,7,130061 +3,64,1,8,146999 +3,64,1,9,133150 +3,64,1,10,128341 +3,64,2,1,108638 +3,64,2,2,107973 +3,64,2,3,108315 +3,64,2,4,107142 +3,64,2,5,107961 +3,64,2,6,105114 +3,64,2,7,111636 +3,64,2,8,118289 +3,64,2,9,110054 +3,64,2,10,107163 +3,256,1,1,480493 +3,256,1,2,470758 +3,256,1,3,476031 +3,256,1,4,540113 +3,256,1,5,479220 +3,256,1,6,478452 +3,256,1,7,479711 +3,256,1,8,511503 +3,256,1,9,560539 +3,256,1,10,552964 +3,256,2,1,388396 +3,256,2,2,392264 +3,256,2,3,393144 +3,256,2,4,418084 +3,256,2,5,393361 +3,256,2,6,388882 +3,256,2,7,387641 +3,256,2,8,387299 +3,256,2,9,387376 +3,256,2,10,387878 +3,512,1,1,935964 +3,512,1,2,940159 +3,512,1,3,936034 +3,512,1,4,1005173 +3,512,1,5,999858 +3,512,1,6,933882 +3,512,1,7,1027540 +3,512,1,8,951639 +3,512,1,9,949518 +3,512,1,10,936740 +3,512,2,1,801546 +3,512,2,2,776749 +3,512,2,3,778238 +3,512,2,4,810279 +3,512,2,5,817510 +3,512,2,6,777993 +3,512,2,7,822376 +3,512,2,8,790479 +3,512,2,9,769155 +3,512,2,10,776583 diff --git a/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-summary.csv b/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-summary.csv new file mode 100644 index 0000000..5e9c4b6 --- /dev/null +++ b/kv-service/benchmarks/results/2026-10-02-skv-x86-rail-mock-summary.csv @@ -0,0 +1,19 @@ +trial,size_mib,rails,iterations,mean_read_us,effective_gib_s,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes +1,64,1,10,131854,0.474,648647,670107,263224,671088640,0 +1,64,2,10,110443,0.566,751398,668147,263216,335544320,335544320 +1,256,1,10,495271,0.505,2522735,2430084,1049680,2684354560,0 +1,256,2,10,410433,0.609,2620414,2538114,1049764,1342177280,1342177280 +1,512,1,10,977063,0.512,4924127,4846235,2098236,5368709120,0 +1,512,2,10,841624,0.594,5631231,4943625,2098416,2684354560,2684354560 +2,64,1,10,147548,0.424,705049,770597,263160,671088640,0 +2,64,2,10,109598,0.57,696259,706594,263132,335544320,335544320 +2,256,1,10,493835,0.506,2662242,2276306,1049672,2684354560,0 +2,256,2,10,465536,0.537,2833254,3176520,1049708,1342177280,1342177280 +2,512,1,10,970911,0.515,4960711,4748029,2098296,5368709120,0 +2,512,2,10,791761,0.632,5182862,4737840,2098312,2684354560,2684354560 +3,64,1,10,131052,0.477,678959,631675,263064,671088640,0 +3,64,2,10,109228,0.572,654995,747248,263152,335544320,335544320 +3,256,1,10,502978,0.497,2459817,2569881,1049740,2684354560,0 +3,256,2,10,392432,0.637,2488310,2460885,1049692,1342177280,1342177280 +3,512,1,10,961650,0.52,4965114,4650760,2098228,5368709120,0 +3,512,2,10,792090,0.631,4933685,4994187,2098292,2684354560,2684354560 From edb1abc19de8955b65f7b56cbb0e12200dd76d11 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 13:11:29 +0800 Subject: [PATCH 10/20] Exercise two independent RXE rails and bound subset CQ wait Add a fault hook scoped to one RDMA listener and a dual-rail late-WRITE regression using real Verbs. Make stripe-subset CQ polling deadline configurable, preserving uncertain source memory until QP teardown. Add a paired real-Verbs benchmark collector and document Soft-RoCE limitations. --- README.md | 14 +- kv-service/benchmarks/collect_rail_verbs.py | 167 ++++++++++++++++++++ kv-service/client-rs/tests/rail_read_e2e.rs | 43 +++++ kv-service/server/src/rdma/qp.rs | 18 ++- kv-service/server/src/rdma/server.rs | 52 +++++- 5 files changed, 285 insertions(+), 9 deletions(-) create mode 100644 kv-service/benchmarks/collect_rail_verbs.py diff --git a/README.md b/README.md index 45c9b8d..171ed01 100644 --- a/README.md +++ b/README.md @@ -353,13 +353,17 @@ uncertain completion retains its source and MR until its QP is destroyed. For hardware-independent scheduling and failure checks, run `cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma rail_read::tests`. The ignored `rail_read_e2e` suite contains five single -real-Rail checks and three dual-Rail checks. Select a test with `--ignored +real-Rail checks and four dual-Rail checks. Select a test with `--ignored --exact --nocapture` and configure `CS_RAIL_COORDINATOR`, `CS_RAIL_LISTENER0/1`, `CS_RAIL_DEVICE0/1`, and `CS_RAIL_GID0/1` as needed. `CS_RDMA_SLAB_MB=0` on the server exercises the registered-buffer fallback. On an isolated server only, `CS_RDMA_TEST_PRE_WRITE_DELAY_MS=3000` delays stripe-subset WRITEs for the late-completion test; the delay is bounded to -five seconds and must be unset after that test. A physical-stripe corruption +five seconds. Add `CS_RDMA_TEST_PRE_WRITE_NIC_IDX=1` to delay only the second +listener and exercise partial completion. `CS_RDMA_CQ_TIMEOUT_MS` bounds the +stripe-subset CQ poll between 100 and 30,000 ms (default 30,000); use 2,000 +ms for isolated late-WRITE injection. Unset both fault-injection variables +after testing. A physical-stripe corruption test additionally requires checksum verification enabled before writing its object and a reversible fault injection into one test-only stripe file. The ignored `software_only_mock_benchmark` exercises scheduling and memory @@ -374,6 +378,12 @@ that Mock results are host-specific software measurements. The separate `kv-service/benchmarks/results/2026-10-02-skv-single-hca.json` records physical single-rail Verbs reads, HCA port-counter deltas, fault tests, and the storage/topology boundary. It does not measure two-rail aggregation. +Use `kv-service/configs/server-rail-validation.toml` as a small, checksummed +server example. For real Verbs performance, +`kv-service/benchmarks/collect_rail_verbs.py` pairs one +and two rails on the same prewritten objects and saves every latency sample, +CPU time, RSS, and per-rail bytes. Label RXE runs `soft-roce`: RXE executes the +Verbs path but does not establish HCA offload or physical link aggregation. --- diff --git a/kv-service/benchmarks/collect_rail_verbs.py b/kv-service/benchmarks/collect_rail_verbs.py new file mode 100644 index 0000000..a942b31 --- /dev/null +++ b/kv-service/benchmarks/collect_rail_verbs.py @@ -0,0 +1,167 @@ +from __future__ import annotations + +# Capture paired single/dual-rail real Verbs reads without changing object layout. +# The environment label must say whether the devices are physical HCA or RXE. + +import argparse +import csv +import json +import platform +import re +import subprocess +from pathlib import Path + + +SAMPLE = re.compile( + r"^sample,environment=([^,]+),rails=(\d+),iteration=(\d+)," + r"bytes=(\d+),latency_us=(\d+),gib_per_s=([0-9.]+),xxh3=([0-9a-f]+)$", + re.MULTILINE, +) +SUMMARY = re.compile( + r"^summary,environment=([^,]+),rails=(\d+),bytes_per_iter=(\d+)," + r"iters=(\d+),median_us=(\d+),cpu_user_us=(\d+),cpu_system_us=(\d+)," + r"peak_rss_kb=(\d+),rail_bytes=\[([^]]+)\]$", + re.MULTILINE, +) + + +def write_csv(path: Path, rows: list[dict[str, object]]) -> None: + with path.open("w", newline="") as handle: + writer = csv.DictWriter(handle, fieldnames=rows[0].keys(), lineterminator="\n") + writer.writeheader() + writer.writerows(rows) + + +def main() -> None: + parser = argparse.ArgumentParser(description="Collect paired real-Verbs rail reads") + parser.add_argument("--binary", required=True, type=Path) + parser.add_argument("--environment", required=True, choices=["physical", "soft-roce"]) + parser.add_argument("--coordinator", required=True) + parser.add_argument("--namespace", required=True) + parser.add_argument("--object", action="append", required=True, help="SIZE_MIB:OBJECT_KEY") + parser.add_argument("--rail", action="append", required=True, help="SDK rail specification") + parser.add_argument("--trials", type=int, default=3) + parser.add_argument("--iterations", type=int, default=5) + parser.add_argument("--warmup", type=int, default=1) + parser.add_argument("--output-dir", required=True, type=Path) + parser.add_argument("--output-prefix", default="rail-verbs") + parser.add_argument("--source-commit", required=True) + parser.add_argument("--notes", default="") + args = parser.parse_args() + if len(args.rail) != 2 or args.trials <= 0 or args.iterations <= 0: + parser.error("two --rail entries and positive trials/iterations are required") + objects = [] + for spec in args.object: + size_text, separator, key = spec.partition(":") + if not separator or not key or not size_text.isdigit() or int(size_text) <= 0: + parser.error("each --object must be SIZE_MIB:OBJECT_KEY") + objects.append((int(size_text), key)) + + args.output_dir.mkdir(parents=True, exist_ok=True) + summaries: list[dict[str, object]] = [] + samples: list[dict[str, object]] = [] + expected_hashes: dict[int, str] = {} + for trial in range(1, args.trials + 1): + for size_mib, key in objects: + for rails in (1, 2): + command = [ + str(args.binary), + "--environment", + args.environment, + "--coordinator", + args.coordinator, + "--namespace", + args.namespace, + "--object-key", + key, + "--warmup", + str(args.warmup), + "--iterations", + str(args.iterations), + ] + for rail in args.rail[:rails]: + command.extend(["--rail", rail]) + output = subprocess.run( + command, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + check=True, + ).stdout + sample_matches = SAMPLE.findall(output) + summary_match = SUMMARY.search(output) + if len(sample_matches) != args.iterations or summary_match is None: + raise RuntimeError(f"incomplete benchmark output: {output[-2000:]}") + run_hashes = {row[6] for row in sample_matches} + if len(run_hashes) != 1: + raise RuntimeError("object hash changed within one benchmark run") + checksum = run_hashes.pop() + if size_mib in expected_hashes and expected_hashes[size_mib] != checksum: + raise RuntimeError("single- and dual-rail hashes disagree") + expected_hashes[size_mib] = checksum + for _, _, iteration, byte_count, latency_us, gib_per_s, _ in sample_matches: + if int(byte_count) != size_mib * 1024 * 1024: + raise RuntimeError("benchmark returned an unexpected object size") + samples.append( + dict( + trial=trial, + size_mib=size_mib, + rails=rails, + iteration=int(iteration), + bytes=int(byte_count), + latency_us=int(latency_us), + gib_per_s=float(gib_per_s), + xxh3=checksum, + ) + ) + _, _, _, _, median_us, cpu_user, cpu_system, rss, rail_text = summary_match.groups() + rail_bytes = [int(value) for value in rail_text.split(",")] + if ( + len(rail_bytes) != rails + or sum(rail_bytes) != size_mib * 1024 * 1024 * args.iterations + ): + raise RuntimeError("per-rail counters disagree with total completed bytes") + if any(value == 0 for value in rail_bytes): + raise RuntimeError("one configured rail transferred no object bytes") + summaries.append( + dict( + trial=trial, + size_mib=size_mib, + rails=rails, + iterations=args.iterations, + median_us=int(median_us), + cpu_user_us=int(cpu_user), + cpu_system_us=int(cpu_system), + peak_rss_kb=int(rss), + rail0_bytes=rail_bytes[0], + rail1_bytes=rail_bytes[1] if rails == 2 else 0, + xxh3=checksum, + ) + ) + print( + f"trial={trial} size={size_mib}MiB rails={rails} median_us={median_us}", + flush=True, + ) + + write_csv(args.output_dir / f"{args.output_prefix}-summary.csv", summaries) + write_csv(args.output_dir / f"{args.output_prefix}-samples.csv", samples) + environment = { + "environment": args.environment, + "source_commit": args.source_commit, + "platform": platform.platform(), + "machine": platform.machine(), + "notes": args.notes, + "objects_mib": [size for size, _ in objects], + "trials": args.trials, + "iterations_per_trial": args.iterations, + "warmup_reads": args.warmup, + "coordinator": args.coordinator, + "rails": args.rail, + } + (args.output_dir / f"{args.output_prefix}-environment.json").write_text( + json.dumps(environment, indent=2) + "\n" + ) + + +if __name__ == "__main__": + main() diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs index 246fbb8..a3a37d6 100644 --- a/kv-service/client-rs/tests/rail_read_e2e.rs +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -252,6 +252,49 @@ async fn late_server_write_after_timeout_cannot_corrupt_reused_buffer() { assert!(destination.iter().all(|byte| *byte == 0x33)); } +#[tokio::test] +#[ignore = "requires two real rails and a three-second pre-WRITE delay on server nic_idx=1"] +async fn late_second_rail_after_first_completion_cannot_publish_or_corrupt() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener0 = setting("CS_RAIL_LISTENER0", &advertised); + let listener1 = setting("CS_RAIL_LISTENER1", "127.0.0.1:50054"); + let reader = Arc::new( + RailReader::new( + vec![ + route(&advertised, 0, &listener0), + route(&advertised, 1, &listener1), + ], + RailLimits { + io_timeout: Duration::from_secs(1), + ..RailLimits::default() + }, + ) + .expect("dual reader"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert!(client + .read_multi_rail_into( + Arc::clone(&reader), + "rail-e2e", + &key, + &mut destination, + None, + ) + .await + .is_err()); + let snapshots = reader.snapshots(); + assert_eq!(snapshots[0].reads_ok, 1, "first rail must have completed"); + assert_eq!(snapshots[1].reads_err, 1, "second rail must have failed"); + assert!(destination.iter().all(|byte| *byte == 0xA5)); + destination.fill(0x33); + tokio::time::sleep(Duration::from_secs(8)).await; + assert!(destination.iter().all(|byte| *byte == 0x33)); + assert!(reader + .snapshots() + .iter() + .all(|snapshot| snapshot.inflight_requests == 0 && snapshot.registered_bytes == 0)); +} + #[tokio::test] #[ignore = "requires two reachable RDMA listeners on the same storage node"] async fn same_object_matches_single_and_dual_rail() { diff --git a/kv-service/server/src/rdma/qp.rs b/kv-service/server/src/rdma/qp.rs index 4a19376..905a60c 100644 --- a/kv-service/server/src/rdma/qp.rs +++ b/kv-service/server/src/rdma/qp.rs @@ -252,6 +252,15 @@ impl RcQp { /// Wait for N work completions. Simple busy poll for PoC use. pub fn poll_n(cq: NonNull, expected: usize) -> Result<()> { + Self::poll_n_timeout(cq, expected, std::time::Duration::from_secs(30)) + } + + /// Wait for N work completions within a caller-selected deadline. + pub fn poll_n_timeout( + cq: NonNull, + expected: usize, + timeout: std::time::Duration, + ) -> Result<()> { unsafe { // ibv_wc does not implement Clone; use push instead of vec!. let mut wcs: Vec = Vec::with_capacity(expected.max(1)); @@ -285,8 +294,13 @@ impl RcQp { } got += n as usize; } - if start.elapsed().as_secs() > 30 { - return Err(anyhow!("poll_n timeout after 30s, got {}/{}", got, expected)); + if start.elapsed() >= timeout { + return Err(anyhow!( + "poll_n timeout after {}ms, got {}/{}", + timeout.as_millis(), + got, + expected + )); } } match first_error { diff --git a/kv-service/server/src/rdma/server.rs b/kv-service/server/src/rdma/server.rs index 4291bf2..859715c 100644 --- a/kv-service/server/src/rdma/server.rs +++ b/kv-service/server/src/rdma/server.rs @@ -1119,6 +1119,18 @@ fn ensure_subset_complete(expected_bytes: usize, actual_bytes: u64) -> Result<() Ok(()) } +fn parse_subset_cq_timeout(raw_ms: Option<&str>) -> std::time::Duration { + let millis = raw_ms + .and_then(|raw| raw.parse::().ok()) + .unwrap_or(30_000) + .clamp(100, 30_000); + std::time::Duration::from_millis(millis) +} + +fn subset_cq_timeout() -> std::time::Duration { + parse_subset_cq_timeout(std::env::var("CS_RDMA_CQ_TIMEOUT_MS").ok().as_deref()) +} + fn serve_get_stripes_fallback( kv_ctx: &Arc, rdma: &Arc, @@ -1127,6 +1139,7 @@ fn serve_get_stripes_fallback( striping: &StripingInfo, req: &DescriptorGetReqMsg, ) -> Result<(bool, u64, u32)> { + let cq_timeout = subset_cq_timeout(); let indices = req .stripes .iter() @@ -1175,7 +1188,7 @@ fn serve_get_stripes_fallback( length, true, )?; - if let Err(error) = RcQp::poll_n(client_cq, 1) { + if let Err(error) = RcQp::poll_n_timeout(client_cq, 1, cq_timeout) { qp.retain_uncertain_write(mr, segment); return Err(error); } @@ -1311,10 +1324,15 @@ fn serve_get_stripes( // its QP/MR before this server attempts a late WRITE. Unset in production. if let Ok(raw_delay) = std::env::var("CS_RDMA_TEST_PRE_WRITE_DELAY_MS") { if let Ok(delay_ms) = raw_delay.parse::() { - if delay_ms > 0 { + let nic_matches = match std::env::var("CS_RDMA_TEST_PRE_WRITE_NIC_IDX") { + Ok(raw_nic) => raw_nic.parse::().ok() == Some(nic_idx), + Err(_) => true, + }; + if delay_ms > 0 && nic_matches { let bounded_ms = delay_ms.min(5_000); tracing::warn!( event = "rdma_test_pre_write_delay", + nic_idx, delay_ms = bounded_ms, "delaying stripe-subset WRITEs for fault injection" ); @@ -1324,6 +1342,7 @@ fn serve_get_stripes( } let view = extent.view(nic_idx); const COMPLETION_WINDOW: usize = RcQp::MAX_SEND_WR / 2; + let cq_timeout = subset_cq_timeout(); let mut outstanding = 0usize; let mut writes_posted = 0usize; let mut total = 0u64; @@ -1402,7 +1421,7 @@ fn serve_get_stripes( } } while first_error.is_none() && outstanding >= COMPLETION_WINDOW { - match RcQp::poll_n(client_cq, 1) { + match RcQp::poll_n_timeout(client_cq, 1, cq_timeout) { Ok(()) => outstanding -= 1, Err(error) => { uncertain_write = true; @@ -1428,7 +1447,7 @@ fn serve_get_stripes( if outstanding > 0 && !uncertain_write { let poll_start = std::time::Instant::now(); - let poll_result = RcQp::poll_n(client_cq, outstanding); + let poll_result = RcQp::poll_n_timeout(client_cq, outstanding, cq_timeout); poll_us += poll_start.elapsed().as_micros() as u64; if let Err(error) = poll_result { uncertain_write = true; @@ -2105,7 +2124,10 @@ mod tests { #[cfg(test)] mod sge_tests { - use super::{ensure_subset_complete, fallback_write_targets, map_range_to_segments}; + use super::{ + ensure_subset_complete, fallback_write_targets, map_range_to_segments, + parse_subset_cq_timeout, + }; #[test] fn range_within_one_segment() { @@ -2159,4 +2181,24 @@ mod sge_tests { assert!(ensure_subset_complete(64, 128).is_err()); assert!(ensure_subset_complete(64, 64).is_ok()); } + + #[test] + fn cq_timeout_configuration_is_bounded() { + assert_eq!( + parse_subset_cq_timeout(None), + std::time::Duration::from_secs(30) + ); + assert_eq!( + parse_subset_cq_timeout(Some("2000")), + std::time::Duration::from_secs(2) + ); + assert_eq!( + parse_subset_cq_timeout(Some("0")), + std::time::Duration::from_millis(100) + ); + assert_eq!( + parse_subset_cq_timeout(Some("999999")), + std::time::Duration::from_secs(30) + ); + } } From 52b933d97cb00ddd38d5fa93dee8ec5bebb8eb08 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 13:13:38 +0800 Subject: [PATCH 11/20] Alternate rail order in paired Verbs benchmarks Reverse one- and two-rail execution order on even trials to reduce a systematic warm-cache or host-load bias in the reported software RoCE matrix. --- kv-service/benchmarks/collect_rail_verbs.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kv-service/benchmarks/collect_rail_verbs.py b/kv-service/benchmarks/collect_rail_verbs.py index a942b31..c8c9bd2 100644 --- a/kv-service/benchmarks/collect_rail_verbs.py +++ b/kv-service/benchmarks/collect_rail_verbs.py @@ -63,7 +63,8 @@ def main() -> None: expected_hashes: dict[int, str] = {} for trial in range(1, args.trials + 1): for size_mib, key in objects: - for rails in (1, 2): + # Alternate order to reduce a systematic warm-cache advantage. + for rails in (1, 2) if trial % 2 else (2, 1): command = [ str(args.binary), "--environment", From cd529b53759281971d1af121db4bccb6d0ef6220 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 13:23:26 +0800 Subject: [PATCH 12/20] Measure concurrent object reads through one shared RailReader Add an opt-in CLI concurrency mode that keeps full reads in one process, joins every request, checks object hashes, and reports batch throughput, CPU, RSS, and per-rail bytes. Add a paired real-Verbs collector for concurrency cells above one. --- README.md | 5 + .../benchmarks/collect_rail_concurrent.py | 215 ++++++++++++++++++ .../client-rs/src/bin/rail_read_bench.rs | 132 ++++++++++- 3 files changed, 349 insertions(+), 3 deletions(-) create mode 100644 kv-service/benchmarks/collect_rail_concurrent.py diff --git a/README.md b/README.md index 171ed01..676c064 100644 --- a/README.md +++ b/README.md @@ -384,6 +384,11 @@ server example. For real Verbs performance, and two rails on the same prewritten objects and saves every latency sample, CPU time, RSS, and per-rail bytes. Label RXE runs `soft-roce`: RXE executes the Verbs path but does not establish HCA offload or physical link aggregation. +The CLI's `--concurrency N` option (1–8) runs N complete object reads within +one process and one shared RailReader, reporting per-request latency and +aggregate throughput. `kv-service/benchmarks/collect_rail_concurrent.py` +collects paired 1/2-rail batches for concurrency levels above one; it preserves +the same object, server, layout, and concurrency within each comparison. --- diff --git a/kv-service/benchmarks/collect_rail_concurrent.py b/kv-service/benchmarks/collect_rail_concurrent.py new file mode 100644 index 0000000..8e27f01 --- /dev/null +++ b/kv-service/benchmarks/collect_rail_concurrent.py @@ -0,0 +1,215 @@ +from __future__ import annotations + +# Measure concurrent reads within one Rust Worker/RailReader over real Verbs. +# Compare one/two rails only within the same object size and concurrency cell. + +import argparse +import csv +import json +import platform +import re +import subprocess +from pathlib import Path + + +SAMPLE = re.compile( + r"^sample,environment=([^,]+),rails=(\d+),concurrency=(\d+)," + r"batch=(\d+),worker=(\d+),bytes=(\d+),latency_us=(\d+),xxh3=([0-9a-f]+)$", + re.MULTILINE, +) +BATCH = re.compile( + r"^batch,environment=([^,]+),rails=(\d+),concurrency=(\d+)," r"iteration=(\d+),wall_us=(\d+)$", + re.MULTILINE, +) +SUMMARY = re.compile( + r"^summary_concurrent,environment=([^,]+),rails=(\d+),concurrency=(\d+)," + r"bytes_per_read=(\d+),batches=(\d+),median_request_us=(\d+)," + r"median_batch_us=(\d+),aggregate_gib_per_s=([0-9.]+)," + r"cpu_user_us=(\d+),cpu_system_us=(\d+),peak_rss_kb=(\d+)," + r"rail_bytes=\[([^]]+)\]$", + re.MULTILINE, +) + + +def write_csv(path: Path, rows: list[dict[str, object]]) -> None: + with path.open("w", newline="") as handle: + writer = csv.DictWriter(handle, fieldnames=rows[0].keys(), lineterminator="\n") + writer.writeheader() + writer.writerows(rows) + + +def main() -> None: + parser = argparse.ArgumentParser(description="Collect concurrent real-Verbs rail reads") + parser.add_argument("--binary", required=True, type=Path) + parser.add_argument("--environment", required=True, choices=["physical", "soft-roce"]) + parser.add_argument("--coordinator", required=True) + parser.add_argument("--namespace", required=True) + parser.add_argument("--object", action="append", required=True, help="SIZE_MIB:OBJECT_KEY") + parser.add_argument("--rail", action="append", required=True) + parser.add_argument("--concurrency", action="append", type=int, required=True) + parser.add_argument("--trials", type=int, default=2) + parser.add_argument("--batches", type=int, default=3) + parser.add_argument("--warmup", type=int, default=1) + parser.add_argument("--output-dir", required=True, type=Path) + parser.add_argument("--output-prefix", default="rail-verbs-concurrent") + parser.add_argument("--source-commit", required=True) + parser.add_argument("--notes", default="") + args = parser.parse_args() + if len(args.rail) != 2 or args.trials <= 0 or args.batches <= 0: + parser.error("two --rail entries and positive trials/batches are required") + if any(value < 2 or value > 8 for value in args.concurrency): + parser.error("concurrency values must be in 2..=8") + objects = [] + for spec in args.object: + size_text, separator, key = spec.partition(":") + if not separator or not key or not size_text.isdigit() or int(size_text) <= 0: + parser.error("each --object must be SIZE_MIB:OBJECT_KEY") + objects.append((int(size_text), key)) + + args.output_dir.mkdir(parents=True, exist_ok=True) + summaries: list[dict[str, object]] = [] + samples: list[dict[str, object]] = [] + batches: list[dict[str, object]] = [] + expected_hashes: dict[int, str] = {} + for trial in range(1, args.trials + 1): + for size_mib, key in objects: + for concurrency in args.concurrency: + for rails in (1, 2) if trial % 2 else (2, 1): + command = [ + str(args.binary), + "--environment", + args.environment, + "--coordinator", + args.coordinator, + "--namespace", + args.namespace, + "--object-key", + key, + "--warmup", + str(args.warmup), + "--iterations", + str(args.batches), + "--concurrency", + str(concurrency), + ] + for rail in args.rail[:rails]: + command.extend(["--rail", rail]) + output = subprocess.run( + command, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + check=True, + ).stdout + sample_matches = SAMPLE.findall(output) + batch_matches = BATCH.findall(output) + summary_match = SUMMARY.search(output) + if ( + len(sample_matches) != args.batches * concurrency + or len(batch_matches) != args.batches + or summary_match is None + ): + raise RuntimeError(f"incomplete concurrent output: {output[-2000:]}") + hashes = {row[7] for row in sample_matches} + if len(hashes) != 1: + raise RuntimeError("concurrent readers returned different objects") + checksum = hashes.pop() + if size_mib in expected_hashes and expected_hashes[size_mib] != checksum: + raise RuntimeError("one- and two-rail hashes disagree") + expected_hashes[size_mib] = checksum + for _, _, _, batch, worker, byte_count, latency_us, _ in sample_matches: + if int(byte_count) != size_mib * 1024 * 1024: + raise RuntimeError("unexpected object size") + samples.append( + dict( + trial=trial, + size_mib=size_mib, + rails=rails, + concurrency=concurrency, + batch=int(batch), + worker=int(worker), + bytes=int(byte_count), + latency_us=int(latency_us), + xxh3=checksum, + ) + ) + for _, _, _, iteration, wall_us in batch_matches: + batches.append( + dict( + trial=trial, + size_mib=size_mib, + rails=rails, + concurrency=concurrency, + batch=int(iteration), + wall_us=int(wall_us), + ) + ) + ( + _, + _, + _, + _, + _, + median_request, + median_batch, + throughput, + cpu_user, + cpu_system, + rss, + rail_text, + ) = summary_match.groups() + rail_bytes = [int(value) for value in rail_text.split(",")] + if len(rail_bytes) != rails or sum(rail_bytes) != ( + size_mib * 1024 * 1024 * args.batches * concurrency + ): + raise RuntimeError("per-rail byte counters disagree") + if any(value == 0 for value in rail_bytes): + raise RuntimeError("one configured rail transferred no bytes") + summaries.append( + dict( + trial=trial, + size_mib=size_mib, + rails=rails, + concurrency=concurrency, + batches=args.batches, + median_request_us=int(median_request), + median_batch_us=int(median_batch), + aggregate_gib_per_s=float(throughput), + cpu_user_us=int(cpu_user), + cpu_system_us=int(cpu_system), + peak_rss_kb=int(rss), + rail0_bytes=rail_bytes[0], + rail1_bytes=rail_bytes[1] if rails == 2 else 0, + xxh3=checksum, + ) + ) + print( + f"trial={trial} size={size_mib}MiB concurrency={concurrency} " + f"rails={rails} aggregate_gib_s={throughput}", + flush=True, + ) + + write_csv(args.output_dir / f"{args.output_prefix}-summary.csv", summaries) + write_csv(args.output_dir / f"{args.output_prefix}-samples.csv", samples) + write_csv(args.output_dir / f"{args.output_prefix}-batches.csv", batches) + environment = { + "environment": args.environment, + "source_commit": args.source_commit, + "platform": platform.platform(), + "machine": platform.machine(), + "notes": args.notes, + "objects_mib": [size for size, _ in objects], + "concurrency": args.concurrency, + "trials": args.trials, + "batches_per_trial": args.batches, + "warmup_reads": args.warmup, + "coordinator": args.coordinator, + "rails": args.rail, + } + (args.output_dir / f"{args.output_prefix}-environment.json").write_text( + json.dumps(environment, indent=2) + "\n" + ) + + +if __name__ == "__main__": + main() diff --git a/kv-service/client-rs/src/bin/rail_read_bench.rs b/kv-service/client-rs/src/bin/rail_read_bench.rs index 5bf28c7..29a8e4e 100644 --- a/kv-service/client-rs/src/bin/rail_read_bench.rs +++ b/kv-service/client-rs/src/bin/rail_read_bench.rs @@ -6,7 +6,7 @@ use contextstore_client_rs::rail_read::{RailLimits, RailReader, RailRoute}; use contextstore_client_rs::rdma::RdmaClientConfig; use contextstore_client_rs::KvClient; use std::sync::Arc; -use std::time::Instant; +use std::time::{Duration, Instant}; #[derive(Parser)] #[command(about = "Read one ContextStore object over independently configured RDMA rails")] @@ -27,6 +27,9 @@ struct Args { warmup: usize, #[arg(long, default_value_t = 5)] iterations: usize, + /// Concurrent object reads within this Worker, sharing one RailReader. + #[arg(long, default_value_t = 1)] + concurrency: usize, } fn parse_route(spec: &str) -> Result { @@ -76,11 +79,131 @@ fn process_usage() -> (u64, u64, u64) { (0, 0, 0) } +async fn run_concurrent( + args: &Args, + client: &KvClient, + reader: Arc, + size: usize, + warmup_bytes: &[u64], +) -> Result<()> { + let mut request_times = Vec::::new(); + let mut batch_times = Vec::::new(); + let mut cpu_user_us = 0u64; + let mut cpu_system_us = 0u64; + let mut expected_hash = None; + for batch in 0..args.iterations { + let destinations = (0..args.concurrency) + .map(|_| vec![0xA5; size]) + .collect::>(); + let before_cpu = process_usage(); + let batch_started = Instant::now(); + let mut handles = Vec::with_capacity(args.concurrency); + for (worker, mut destination) in destinations.into_iter().enumerate() { + let mut worker_client = client.clone(); + let worker_reader = Arc::clone(&reader); + let namespace = args.namespace.clone(); + let object_key = args.object_key.clone(); + handles.push(tokio::spawn(async move { + let started = Instant::now(); + let result = worker_client + .read_multi_rail_into( + worker_reader, + &namespace, + &object_key, + &mut destination, + None, + ) + .await; + (worker, result, destination, started.elapsed()) + })); + } + let mut first_error = None; + for handle in handles { + let (worker, result, destination, elapsed) = handle.await?; + match result { + Ok(Some(bytes)) if bytes == size => { + let checksum = twox_hash::xxh3::hash64(&destination); + if expected_hash.is_some_and(|expected| expected != checksum) { + first_error.get_or_insert_with(|| anyhow!("object hash changed")); + } + expected_hash.get_or_insert(checksum); + println!( + "sample,environment={},rails={},concurrency={},batch={},worker={},bytes={},latency_us={},xxh3={checksum:016x}", + args.environment, + reader.snapshots().len(), + args.concurrency, + batch + 1, + worker, + bytes, + elapsed.as_micros() + ); + request_times.push(elapsed); + } + Ok(Some(bytes)) => { + first_error.get_or_insert_with(|| anyhow!("short read: {bytes} of {size}")); + } + Ok(None) => { + first_error.get_or_insert_with(|| anyhow!("object disappeared")); + } + Err(error) => { + first_error.get_or_insert(error); + } + } + } + let wall = batch_started.elapsed(); + let after_cpu = process_usage(); + cpu_user_us += after_cpu.0 - before_cpu.0; + cpu_system_us += after_cpu.1 - before_cpu.1; + if let Some(error) = first_error { + return Err(error); + } + println!( + "batch,environment={},rails={},concurrency={},iteration={},wall_us={}", + args.environment, + reader.snapshots().len(), + args.concurrency, + batch + 1, + wall.as_micros() + ); + batch_times.push(wall); + } + request_times.sort_unstable(); + batch_times.sort_unstable(); + let total_wall_seconds: f64 = batch_times.iter().map(Duration::as_secs_f64).sum(); + let total_bytes = size + .checked_mul(args.concurrency) + .and_then(|bytes| bytes.checked_mul(args.iterations)) + .ok_or_else(|| anyhow!("benchmark byte count overflow"))?; + let aggregate_gib_per_s = total_bytes as f64 / total_wall_seconds / 1024f64.powi(3); + let rail_bytes: Vec<_> = reader + .snapshots() + .iter() + .zip(warmup_bytes) + .map(|(snapshot, warmup)| snapshot.bytes - warmup) + .collect(); + println!( + "summary_concurrent,environment={},rails={},concurrency={},bytes_per_read={},batches={},median_request_us={},median_batch_us={},aggregate_gib_per_s={aggregate_gib_per_s:.3},cpu_user_us={},cpu_system_us={},peak_rss_kb={},rail_bytes={rail_bytes:?}", + args.environment, + rail_bytes.len(), + args.concurrency, + size, + args.iterations, + request_times[request_times.len() / 2].as_micros(), + batch_times[batch_times.len() / 2].as_micros(), + cpu_user_us, + cpu_system_us, + process_usage().2 + ); + Ok(()) +} + #[tokio::main] async fn main() -> Result<()> { let args = Args::parse(); - if args.iterations == 0 { - return Err(anyhow!("--iterations must be positive")); + if args.iterations == 0 || !(1..=8).contains(&args.concurrency) { + return Err(anyhow!( + "--iterations must be positive and --concurrency must be 1..=8" + )); } let routes = args .rails @@ -128,6 +251,9 @@ async fn main() -> Result<()> { .ok_or_else(|| anyhow!("object disappeared during warmup"))?; } let warmup_bytes: Vec<_> = reader.snapshots().iter().map(|rail| rail.bytes).collect(); + if args.concurrency > 1 { + return run_concurrent(&args, &client, reader, size, &warmup_bytes).await; + } let mut times = Vec::with_capacity(args.iterations); let mut cpu_user_us = 0u64; let mut cpu_system_us = 0u64; From 0e4c93d779f1375b8e0a77f7c09cccccb21f7fef Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 13:44:21 +0800 Subject: [PATCH 13/20] Cap RDMA registration budget by process memlock limit Reserve 20 percent of a finite Linux RLIMIT_MEMLOCK for runtime headroom and reject excess object reads before QP dispatch. Verify that a 128 MiB by four request on the isolated RXE guest now reports a rail resource limit instead of a late ibv_reg_mr ENOMEM; add real two-rail checksum and cancellation tests. --- README.md | 11 +- kv-service/client-rs/src/rail_read.rs | 34 +++++- kv-service/client-rs/src/rail_read/tests.rs | 8 ++ kv-service/client-rs/tests/rail_read_e2e.rs | 119 ++++++++++++++++++++ 4 files changed, 166 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index 676c064..2f7620e 100644 --- a/README.md +++ b/README.md @@ -343,9 +343,12 @@ publishing share a gate, so whichever starts first determines the result; a failed or cancelled read leaves its caller buffer unchanged. Each rail owns a compact registered receive buffer. The reader bounds concurrent tasks and aggregate in-flight bytes per rail across simultaneous requests. It retains -the active-read and final-staging budget through the post-read lookup and publish step; transfer -and MR reservations end after all rail workers finish. No in-request -transparent retry is attempted. +the active-read and final-staging budget through the post-read lookup and +publish step; transfer and MR reservations end after all rail workers finish. +On Linux, the reader caps its registered-memory budget at 80% of the process's +soft `RLIMIT_MEMLOCK` when finite, rejecting excess reads before dispatch. +Multiple independent RailReader instances still need a deployment-wide +budget. No in-request transparent retry is attempted. The server's stripe-subset fallback also honors tag-15 scatter destinations when its registered slab cannot provide staging space. A fallback WRITE with uncertain completion retains its source and MR until its QP is destroyed. @@ -353,7 +356,7 @@ uncertain completion retains its source and MR until its QP is destroyed. For hardware-independent scheduling and failure checks, run `cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma rail_read::tests`. The ignored `rail_read_e2e` suite contains five single -real-Rail checks and four dual-Rail checks. Select a test with `--ignored +real-Rail checks and six dual-Rail checks. Select a test with `--ignored --exact --nocapture` and configure `CS_RAIL_COORDINATOR`, `CS_RAIL_LISTENER0/1`, `CS_RAIL_DEVICE0/1`, and `CS_RAIL_GID0/1` as needed. `CS_RDMA_SLAB_MB=0` on the server exercises the registered-buffer fallback. diff --git a/kv-service/client-rs/src/rail_read.rs b/kv-service/client-rs/src/rail_read.rs index 7af7d66..4ae199c 100644 --- a/kv-service/client-rs/src/rail_read.rs +++ b/kv-service/client-rs/src/rail_read.rs @@ -595,9 +595,32 @@ impl StagedRailRead { } } +fn effective_registered_budget(configured: u64, soft_memlock_limit: Option) -> u64 { + match soft_memlock_limit { + Some(limit) => configured.min(limit.saturating_sub(limit / 5)), + None => configured, + } +} + +#[cfg(target_os = "linux")] +fn process_memlock_limit() -> Option { + let mut limit: libc::rlimit = unsafe { std::mem::zeroed() }; + if unsafe { libc::getrlimit(libc::RLIMIT_MEMLOCK, &mut limit) } != 0 + || limit.rlim_cur == libc::RLIM_INFINITY + { + return None; + } + Some(limit.rlim_cur) +} + +#[cfg(not(target_os = "linux"))] +fn process_memlock_limit() -> Option { + None +} + impl RailReader { /// Validate routes and create an RDMA reader without opening connections. - pub fn new(routes: Vec, limits: RailLimits) -> Result { + pub fn new(routes: Vec, mut limits: RailLimits) -> Result { if routes.is_empty() || limits.max_active_reads == 0 || limits.max_active_reads_per_rail == 0 @@ -644,6 +667,8 @@ impl RailReader { }) .collect(); let route_count = routes.len(); + limits.max_registered_bytes = + effective_registered_budget(limits.max_registered_bytes, process_memlock_limit()); Ok(Self { routes, topologies, @@ -716,9 +741,14 @@ impl RailReader { .map(|task| (task.route_index, task.packed_len as u64)) .collect(); let mut budget = self.budget.lock().unwrap(); + if budget.registered_bytes.saturating_add(registered) > self.limits.max_registered_bytes { + return Err(RailReadError::ResourceExhausted(format!( + "registered bytes exceed configured or process memlock budget ({} bytes)", + self.limits.max_registered_bytes + ))); + } if budget.active_reads >= self.limits.max_active_reads || budget.staging_bytes.saturating_add(staging) > self.limits.max_staging_bytes - || budget.registered_bytes.saturating_add(registered) > self.limits.max_registered_bytes || budget.inflight_bytes.saturating_add(inflight) > self.limits.max_inflight_bytes { return Err(RailReadError::ResourceExhausted( diff --git a/kv-service/client-rs/src/rail_read/tests.rs b/kv-service/client-rs/src/rail_read/tests.rs index c8b93b5..8aaab75 100644 --- a/kv-service/client-rs/src/rail_read/tests.rs +++ b/kv-service/client-rs/src/rail_read/tests.rs @@ -165,6 +165,14 @@ fn registration_budget_rejects_before_transport() { assert!(mock.calls.lock().unwrap().is_empty()); } +#[test] +fn process_memlock_headroom_caps_registered_bytes_before_dispatch() { + let budget = effective_registered_budget(4 * 1024 * 1024 * 1024, Some(500_012 * 1024)); + assert_eq!(budget, 500_012 * 1024 - (500_012 * 1024) / 5); + assert!(budget < 4 * 128 * 1024 * 1024); + assert_eq!(effective_registered_budget(1024, None), 1024); +} + #[test] fn per_rail_inflight_limit_counts_concurrent_requests() { let (descriptor, placement) = fixture(64, 8); diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs index a3a37d6..9946c33 100644 --- a/kv-service/client-rs/tests/rail_read_e2e.rs +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -189,6 +189,69 @@ async fn cancellation_after_real_transfer_starts_cannot_write_reused_buffer() { assert_eq!(reader.snapshots()[0].registered_bytes, 0); } +#[tokio::test] +#[ignore = "requires two real rails and a large striped object"] +async fn cancellation_during_two_rail_transfer_preserves_reused_buffer() { + let (mut client, key, payload, advertised) = seeded_object().await; + let listener0 = setting("CS_RAIL_LISTENER0", &advertised); + let listener1 = setting("CS_RAIL_LISTENER1", "127.0.0.1:50054"); + let reader = Arc::new( + RailReader::new( + vec![ + route(&advertised, 0, &listener0), + route(&advertised, 1, &listener1), + ], + RailLimits::default(), + ) + .expect("dual reader"), + ); + let cancel = RailCancel::default(); + let task_reader = Arc::clone(&reader); + let task_cancel = cancel.clone(); + let read_task = tokio::spawn(async move { + let mut destination = vec![0xA5; payload.len()]; + let result = client + .read_multi_rail_into( + task_reader, + "rail-e2e", + &key, + &mut destination, + Some(task_cancel), + ) + .await; + (result, destination) + }); + let deadline = Instant::now() + Duration::from_secs(10); + while !reader + .snapshots() + .iter() + .all(|snapshot| snapshot.inflight_requests > 0) + { + assert!( + Instant::now() < deadline, + "both RDMA rails never entered flight" + ); + tokio::time::sleep(Duration::from_millis(1)).await; + } + cancel.cancel(); + let (result, mut destination) = tokio::time::timeout(Duration::from_secs(40), read_task) + .await + .expect("cancelled dual transfer did not quiesce") + .expect("read task joined"); + assert!(result + .expect_err("cancelled dual transfer must fail") + .to_string() + .contains("cancelled")); + assert!(destination.iter().all(|byte| *byte == 0xA5)); + destination.fill(0x33); + tokio::time::sleep(Duration::from_millis(100)).await; + assert!(destination.iter().all(|byte| *byte == 0x33)); + assert!(reader + .snapshots() + .iter() + .all(|snapshot| snapshot.inflight_requests == 0 && snapshot.registered_bytes == 0)); +} + #[tokio::test] #[ignore = "requires an existing striped object with one deliberately corrupted physical stripe"] async fn corrupted_real_stripe_does_not_publish_bytes() { @@ -226,6 +289,62 @@ async fn corrupted_real_stripe_does_not_publish_bytes() { assert!(destination.iter().all(|byte| *byte == 0xA5)); } +#[tokio::test] +#[ignore = "requires two real rails and a deliberately corrupted stripe 1 on an isolated object"] +async fn corrupt_stripe_on_second_rail_cannot_publish_partial_object() { + let coordinator = setting("CS_RAIL_COORDINATOR", "http://127.0.0.1:50051"); + let namespace = setting("CS_RAIL_EXISTING_NAMESPACE", "rust-bench"); + let key = setting("CS_RAIL_EXISTING_KEY", "rail-corrupt0/__combined__"); + let mut client = KvClient::connect(coordinator).await.expect("connect gRPC"); + let lookup = client + .lookup_object(&namespace, &key) + .await + .expect("lookup") + .expect("object exists"); + let placement = lookup.placement.expect("placement"); + assert!( + placement + .chunks + .iter() + .all(|chunk| !chunk.checksum.is_empty()), + "a newly written checksummed object is required" + ); + let advertised = placement.chunks[0].rdma_endpoint.clone(); + let listener0 = setting("CS_RAIL_LISTENER0", &advertised); + let listener1 = setting("CS_RAIL_LISTENER1", "127.0.0.1:50054"); + let reader = Arc::new( + RailReader::new( + vec![ + route(&advertised, 0, &listener0), + route(&advertised, 1, &listener1), + ], + RailLimits::default(), + ) + .expect("dual reader"), + ); + let mut destination = vec![0xA5; lookup.descriptor.size as usize]; + assert!(client + .read_multi_rail_into( + Arc::clone(&reader), + &namespace, + &key, + &mut destination, + None + ) + .await + .is_err()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); + let snapshots = reader.snapshots(); + assert!( + snapshots[0].bytes > 0, + "healthy first rail should have finished" + ); + assert!( + snapshots[1].reads_err > 0 || snapshots[1].bytes > 0, + "second rail must have been involved" + ); +} + #[tokio::test] #[ignore = "requires CS_RDMA_TEST_PRE_WRITE_DELAY_MS=3000 on an isolated real RDMA server"] async fn late_server_write_after_timeout_cannot_corrupt_reused_buffer() { From 8bd364e072ac5a3d538613fc0257edade2eceb01 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 14:15:44 +0800 Subject: [PATCH 14/20] Retire RDMA connections after uncertain completions Stop the server handler after a stripe-subset CQ timeout or failed completion so late CQEs cannot satisfy a later request on the same QP. On client GET control errors, destroy the QP before the caller can deregister or reuse its MR, rejecting later operations on that connection. Reproduce stale response reuse on old binaries and verify the fixed behavior over two RXE rails. --- README.md | 58 ++++++++++++++++- kv-service/client-rs/src/rdma.rs | 72 +++++++++++++-------- kv-service/client-rs/tests/rail_read_e2e.rs | 40 +++++++++++- kv-service/server/src/rdma/server.rs | 54 +++++++++++----- 4 files changed, 181 insertions(+), 43 deletions(-) diff --git a/README.md b/README.md index 2f7620e..0a5de75 100644 --- a/README.md +++ b/README.md @@ -352,11 +352,14 @@ budget. No in-request transparent retry is attempted. The server's stripe-subset fallback also honors tag-15 scatter destinations when its registered slab cannot provide staging space. A fallback WRITE with uncertain completion retains its source and MR until its QP is destroyed. +Uncertain CQ completion terminates the server connection rather than reusing +that CQ for another request; a client GET control error likewise destroys its +QP so a late reply cannot be mistaken for a new request's response. For hardware-independent scheduling and failure checks, run `cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma rail_read::tests`. The ignored `rail_read_e2e` suite contains five single -real-Rail checks and six dual-Rail checks. Select a test with `--ignored +real-Rail checks and seven dual-Rail/control-connection checks. Select a test with `--ignored --exact --nocapture` and configure `CS_RAIL_COORDINATOR`, `CS_RAIL_LISTENER0/1`, `CS_RAIL_DEVICE0/1`, and `CS_RAIL_GID0/1` as needed. `CS_RDMA_SLAB_MB=0` on the server exercises the registered-buffer fallback. @@ -393,6 +396,59 @@ aggregate throughput. `kv-service/benchmarks/collect_rail_concurrent.py` collects paired 1/2-rail batches for concurrency levels above one; it preserves the same object, server, layout, and concurrency within each comparison. +#### Reproduce two independent Soft-RoCE rails in isolated VMs + +The scripts in `kv-service/deploy/softroce-vm/` build two Ubuntu 22.04 KVM +guests on a Linux x86_64 host. They create two test-only tap/bridge pairs with +no production NIC attached. Each guest receives two virtio NICs; four RXE +devices expose separate GIDs and listeners. The VM uses about 8 GiB RAM total +plus sparse qcow2 overlays. Install QEMU/KVM, `genisoimage`, `iproute2`, +`curl`, `tmux`, and Rust/Verbs build dependencies on the host first. + +```bash +export CS_VM_DIR=/absolute/path/to/test-only/vm +export CS_VM_SSH_PUBLIC_KEY=/absolute/path/to/test-key.pub +export CS_VM_TAP_USER="$(id -un)" +kv-service/deploy/softroce-vm/prepare-pair.sh +kv-service/deploy/softroce-vm/setup-host-bridges.sh +kv-service/deploy/softroce-vm/start-pair.sh +``` + +`prepare-pair.sh` verifies Ubuntu's published SHA-256 before boot. Copy +`kv-service/deploy/softroce-vm/ssh_config.example` outside the repository, +replace its KVM host alias and private key path, and set `CS_VM_SSH_CONFIG` to +that file. Run `install-guest-prereqs.sh` and `setup-guest-paths.sh server` +inside the server guest; use `setup-guest-paths.sh client` inside the client +guest. The latter creates `rxe_c0/rxe_c1` and `rxe_s0/rxe_s1` over distinct +`10.31.0.0/24` and `10.32.0.0/24` virtual paths. Check both links with +`rdma link show` and `ping` before running ContextStore. + +Build Linux x86_64 binaries with `build-verbs.sh`. From the Linux build host, +set `CS_VM_SSH_CONFIG` and run `deploy-verbs.sh`; set `CS_E2E_BIN` to the +executable printed by `cargo test --test rail_read_e2e --no-run` if deploying +the ignored hardware tests. The example server config assumes guest user +`railtest`; edit its paths when using another user. Start +`start-guest-server.sh` inside the server guest, then write a striped object +from the client guest with `cs-bench --combined --stream --bytes-pass --only-put`. +Use `cs-rail-read-bench` with one `--rail` and then both rails, or the paired +`collect_rail_verbs.py` script. Every measurement must be labeled `soft-roce`. + +The recorded VM run has [90 sequential samples](kv-service/benchmarks/results/softroce-vm-paired-samples.csv), +[concurrent samples](kv-service/benchmarks/results/softroce-vm-concurrent-64-128-samples.csv), +and [topology/counter evidence](kv-service/benchmarks/results/softroce-vm-topology.json). +`analyze_rail_verbs.py` recomputes descriptive medians and p95 from those +CSVs. For cleanup, stop the test server, run `teardown-guest-paths.sh` in +both guests, power them off, run `stop-pair.sh` if needed, then +`teardown-host-bridges.sh`. These names target only the test-only resources. + +| Deployment / data path | Build and behavior evidence | Performance conclusion | +| --- | --- | --- | +| gRPC-only Rust SDK, no `rdma` feature | Builds without libibverbs; prior Redis two-node integration and Python suites pass | Existing path unchanged | +| One ConnectX-6 Dx HCA rail, Linux x86_64 | Real Verbs object, failure, checksum, late-WRITE and buffer-reuse checks | Single-rail only; test stripes share one SATA SSD | +| Two RXE rails in isolated Ubuntu KVM guests | Real Verbs single/dual reads, generation, checksum, path fault, cancellation, partial completion, late WRITE and registered-memory limits | Software RoCE/QEMU scaling only; no HCA or separate-NVMe claim | +| Two physical HCA rails | Not run: only one HCA port was online on each available SKV node | No physical aggregation claim | +| Linux ARM64 | Earlier RDMA/Mock and server feature suites pass; latest CQ/timeout changes were checked on x86_64 | No ARM64 hardware measurement | + --- ## Deployment shapes diff --git a/kv-service/client-rs/src/rdma.rs b/kv-service/client-rs/src/rdma.rs index f6c6f9a..e9b5ffa 100644 --- a/kv-service/client-rs/src/rdma.rs +++ b/kv-service/client-rs/src/rdma.rs @@ -143,7 +143,7 @@ pub struct RdmaReadOutcome { /// control messages. Create one connection per concurrent transfer worker. pub struct RdmaClient { resources: Arc, - qp: NonNull, + qp: Option>, stream: TcpStream, /// Cached memory registrations keyed by (base pointer, length). /// @@ -361,7 +361,7 @@ impl RdmaClient { match result { Ok(stream) => Ok(Self { resources, - qp, + qp: Some(qp), stream, mr_cache: Vec::new(), }), @@ -372,6 +372,33 @@ impl RdmaClient { } } + fn retire_after_control_error(&mut self) { + let _ = self.stream.shutdown(std::net::Shutdown::Both); + if let Some(qp) = self.qp.take() { + unsafe { ibv_destroy_qp(qp.as_ptr()) }; + } + self.mr_cache.clear(); + } + + fn exchange_get_detailed(&mut self, request: &[u8]) -> Result> { + if self.qp.is_none() { + return Err(anyhow!( + "RDMA connection retired after an earlier control error" + )); + } + let response = (|| { + self.stream.write_all(request)?; + self.stream.flush()?; + read_get_response_detailed(&mut self.stream) + })(); + if response.is_err() { + // A timeout does not prove that the server stopped writing. Tear + // down the QP before the caller may deregister or reuse its MR. + self.retire_after_control_error(); + } + response + } + /// Like [`Self::register_raw_buffer`], but caches the registration inside /// this client keyed by `(ptr, len)` and returns a `Copy` view of it: /// repeated calls with the same region skip `ibv_reg_mr` (~1.5 ms per @@ -490,9 +517,8 @@ impl RdmaClient { rkey, available as u64, )?; - self.stream.write_all(&request)?; - self.stream.flush()?; - read_get_response(&mut self.stream) + self.exchange_get_detailed(&request) + .map(|outcome| outcome.map(|outcome| outcome.bytes)) } /// Read only `stripes` of a striped object into `buffer[offset..]`. @@ -536,9 +562,8 @@ impl RdmaClient { for idx in stripes { request.extend_from_slice(&idx.to_le_bytes()); } - self.stream.write_all(&request)?; - self.stream.flush()?; - read_get_response(&mut self.stream) + self.exchange_get_detailed(&request) + .map(|outcome| outcome.map(|outcome| outcome.bytes)) } /// [`Self::get_descriptor_stripes_into`] for a cached [`BufferView`] @@ -573,9 +598,8 @@ impl RdmaClient { request.extend_from_slice(&idx.to_le_bytes()); } } - self.stream.write_all(&request)?; - self.stream.flush()?; - read_get_response(&mut self.stream) + self.exchange_get_detailed(&request) + .map(|outcome| outcome.map(|outcome| outcome.bytes)) } /// Stripe-subset GET with a scatter destination list (wire tag 15): the @@ -631,9 +655,7 @@ impl RdmaClient { request.extend_from_slice(&rkey.to_le_bytes()); request.extend_from_slice(&len.to_le_bytes()); } - self.stream.write_all(&request)?; - self.stream.flush()?; - read_get_response_detailed(&mut self.stream) + self.exchange_get_detailed(&request) } /// Write `buffer[offset..offset + size]` through the RDMA PUT data path. @@ -831,9 +853,8 @@ impl RdmaClient { ) -> Result { let (dst_addr, rkey, available) = buffer.destination(offset)?; let request = build_get_request(key, dst_addr, rkey, available as u64)?; - self.stream.write_all(&request)?; - self.stream.flush()?; - read_get_response(&mut self.stream) + self.exchange_get_detailed(&request) + .map(|outcome| outcome.map(|outcome| outcome.bytes)) } fn write_remote( @@ -859,7 +880,10 @@ impl RdmaClient { len: len as u32, signaled: index + 1 == writes, }; - post_write(self.qp, &write)?; + let qp = self + .qp + .ok_or_else(|| anyhow!("RDMA connection retired after an earlier control error"))?; + post_write(qp, &write)?; transferred += len; index += 1; } @@ -869,10 +893,10 @@ impl RdmaClient { impl Drop for RdmaClient { fn drop(&mut self) { - let _ = self.stream.write_all(&[MSG_BYE]); - let _ = self.stream.flush(); - unsafe { - ibv_destroy_qp(self.qp.as_ptr()); + if let Some(qp) = self.qp.take() { + let _ = self.stream.write_all(&[MSG_BYE]); + let _ = self.stream.flush(); + unsafe { ibv_destroy_qp(qp.as_ptr()) }; } } } @@ -1190,10 +1214,6 @@ fn read_string(stream: &mut TcpStream, field: &str) -> Result { String::from_utf8(buf).map_err(|e| anyhow!("RDMA {field} utf8: {e}")) } -fn read_get_response(stream: &mut TcpStream) -> Result { - read_get_response_detailed(stream).map(|outcome| outcome.map(|outcome| outcome.bytes)) -} - fn read_get_response_detailed(stream: &mut TcpStream) -> Result> { let mut tag = [0u8; 1]; stream.read_exact(&mut tag)?; diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs index 9946c33..74bc402 100644 --- a/kv-service/client-rs/tests/rail_read_e2e.rs +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -6,7 +6,7 @@ #![cfg(feature = "rdma")] use contextstore_client_rs::rail_read::{RailCancel, RailLimits, RailReader, RailRoute}; -use contextstore_client_rs::rdma::RdmaClientConfig; +use contextstore_client_rs::rdma::{RdmaClient, RdmaClientConfig}; use contextstore_client_rs::KvClient; use prost::bytes::Bytes; use std::sync::Arc; @@ -414,6 +414,44 @@ async fn late_second_rail_after_first_completion_cannot_publish_or_corrupt() { .all(|snapshot| snapshot.inflight_requests == 0 && snapshot.registered_bytes == 0)); } +#[tokio::test] +#[ignore = "requires a three-second delay on server nic_idx=1 and a two-second CQ deadline"] +async fn uncertain_completion_retires_old_server_connection() { + let (mut client, key, _payload, _advertised) = seeded_object().await; + let descriptor = client + .lookup_object("rail-e2e", &key) + .await + .expect("lookup") + .expect("seeded object") + .descriptor; + let device = setting("CS_RAIL_DEVICE1", "rxe_c1"); + let listener = setting("CS_RAIL_LISTENER1", "127.0.0.1:50054"); + let gid = setting("CS_RAIL_GID1", "1") + .parse::() + .expect("GID index"); + let mut rdma = RdmaClient::connect( + RdmaClientConfig::new(listener, device) + .with_gid_index(gid) + .with_io_timeout(Duration::from_secs(1)), + ) + .expect("second rail connects"); + let mut target = vec![0u8; descriptor.size as usize]; + let registered = rdma.register_buffer(&mut target).expect("register buffer"); + let view = registered.view(); + let segments = [(view.addr(), view.rkey(), descriptor.size)]; + assert!(rdma + .get_descriptor_stripes_sge_detailed(&descriptor, &[1], &segments) + .is_err()); + tokio::time::sleep(Duration::from_secs(6)).await; + assert!( + rdma.get_descriptor_stripes_sge_detailed(&descriptor, &[1], &segments) + .is_err(), + "server must close the old QP/CQ rather than return a stale response" + ); + drop(rdma); + drop(registered); +} + #[tokio::test] #[ignore = "requires two reachable RDMA listeners on the same storage node"] async fn same_object_matches_single_and_dual_rail() { diff --git a/kv-service/server/src/rdma/server.rs b/kv-service/server/src/rdma/server.rs index 859715c..c7047cc 100644 --- a/kv-service/server/src/rdma/server.rs +++ b/kv-service/server/src/rdma/server.rs @@ -22,10 +22,9 @@ use crate::rdma::slab::{SlabExtent, SlabPlacement}; use crate::rdma::wire::{ self, DescriptorGetReqMsg, GetRespMsg, PutReadyMsg, PutRespMsg, PutStripeLocation, PutStripesRespMsg, MSG_GET_DESCRIPTOR_REQ, MSG_GET_DESCRIPTOR_STRIPES_REQ, - MSG_GET_DESCRIPTOR_STRIPES_SGE_REQ, MSG_GET_REQ, - MSG_PUT_COMMIT, MSG_PUT_IF_ABSENT_REQ, MSG_PUT_IF_ABSENT_WITH_OPTIONS_REQ, MSG_PUT_REQ, - MSG_PUT_STRIPES_REQ, MSG_PUT_WITH_OPTIONS_REQ, PUT_RESULT_EXISTS, PUT_RESULT_FAILED, - PUT_RESULT_STORED, + MSG_GET_DESCRIPTOR_STRIPES_SGE_REQ, MSG_GET_REQ, MSG_PUT_COMMIT, MSG_PUT_IF_ABSENT_REQ, + MSG_PUT_IF_ABSENT_WITH_OPTIONS_REQ, MSG_PUT_REQ, MSG_PUT_STRIPES_REQ, MSG_PUT_WITH_OPTIONS_REQ, + PUT_RESULT_EXISTS, PUT_RESULT_FAILED, PUT_RESULT_STORED, }; use crate::router::ObjectKey; use crate::KVServiceContext; @@ -1051,7 +1050,6 @@ fn allocate_subset_staging(kv_ctx: &KVServiceContext, size: usize) -> Result Result<() Ok(()) } +fn subset_failure_result(uncertain_completion: bool, error: &str) -> Result<(bool, u64, u32)> { + if uncertain_completion { + // A late CQE must never be consumed by a later request on this QP. + // Propagating an error exits handle_client and destroys the QP/CQ. + Err(anyhow!( + "RDMA connection retired after uncertain stripe-subset completion: {error}" + )) + } else { + Ok((false, 0, 0)) + } +} + fn parse_subset_cq_timeout(raw_ms: Option<&str>) -> std::time::Duration { let millis = raw_ms .and_then(|raw| raw.parse::().ok()) @@ -1274,7 +1284,10 @@ fn serve_get_stripes( ) .is_err() { - tracing::warn!("stripe-subset GET destination does not cover stripe {}", idx); + tracing::warn!( + "stripe-subset GET destination does not cover stripe {}", + idx + ); return Ok((false, 0, 0)); } staged_bytes = staged_bytes @@ -1361,7 +1374,11 @@ fn serve_get_stripes( // tag-15 (SGE): 数据区间按段表映射, 可能拆成多条 WRITE; // tag-12: 单一连续目标, 等价于一个覆盖全对象的段. let targets: Vec<(u64, u32, u64)> = if req.dst_segments.is_empty() { - vec![(req.dst_addr + object_offset as u64, req.dst_rkey, length as u64)] + vec![( + req.dst_addr + object_offset as u64, + req.dst_rkey, + length as u64, + )] } else { match map_range_to_segments( &req.dst_segments, @@ -1392,6 +1409,7 @@ fn serve_get_stripes( first_error = Some(format!( "post RDMA write for stripe {stripe_index}: {error}" )); + uncertain_write = true; post_failed = true; break; } @@ -1410,14 +1428,12 @@ fn serve_get_stripes( outstanding -= n; if let Some(error) = completion_error { uncertain_write = true; - first_error = - Some(format!("poll RDMA completion window: {error}")); + first_error = Some(format!("poll RDMA completion window: {error}")); } } Err(error) => { uncertain_write = true; - first_error = - Some(format!("poll RDMA completion window: {error}")) + first_error = Some(format!("poll RDMA completion window: {error}")) } } while first_error.is_none() && outstanding >= COMPLETION_WINDOW { @@ -1425,8 +1441,7 @@ fn serve_get_stripes( Ok(()) => outstanding -= 1, Err(error) => { uncertain_write = true; - first_error = - Some(format!("poll RDMA completion window: {error}")) + first_error = Some(format!("poll RDMA completion window: {error}")) } } } @@ -1483,7 +1498,7 @@ fn serve_get_stripes( error = %error, "RDMA stripe-subset stream failed without deleting object metadata" ); - return Ok((false, 0, 0)); + return subset_failure_result(uncertain_write, &error); } Ok((true, total, req.stripes.len() as u32)) } @@ -2126,7 +2141,7 @@ mod tests { mod sge_tests { use super::{ ensure_subset_complete, fallback_write_targets, map_range_to_segments, - parse_subset_cq_timeout, + parse_subset_cq_timeout, subset_failure_result, }; #[test] @@ -2201,4 +2216,13 @@ mod sge_tests { std::time::Duration::from_secs(30) ); } + + #[test] + fn uncertain_cq_completion_forces_connection_error() { + assert!(subset_failure_result(true, "CQ timed out").is_err()); + assert_eq!( + subset_failure_result(false, "checksum failure").unwrap(), + (false, 0, 0) + ); + } } From 7f0029e7a4711e84a7d0449b7dd556c6caa28f63 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 14:16:21 +0800 Subject: [PATCH 15/20] Publish reproducible two-rail Soft-RoCE evidence Commit paired 64/128/256 MiB raw Verbs samples, concurrent 2/4 request batches, topology and per-rail virtual-link counters, the bounded memlock/fault test receipts, an analysis script, and isolated KVM deployment scripts. Explicitly label all results as software RXE on one host and one virtual disk. --- kv-service/benchmarks/analyze_rail_verbs.py | 125 ++++++++++ .../results/softroce-vm-analysis.json | 235 ++++++++++++++++++ .../softroce-vm-concurrent-256-batches.csv | 13 + ...oftroce-vm-concurrent-256-environment.json | 21 ++ .../softroce-vm-concurrent-256-samples.csv | 25 ++ .../softroce-vm-concurrent-256-summary.csv | 5 + .../softroce-vm-concurrent-64-128-batches.csv | 49 ++++ ...roce-vm-concurrent-64-128-environment.json | 23 ++ .../softroce-vm-concurrent-64-128-samples.csv | 145 +++++++++++ .../softroce-vm-concurrent-64-128-summary.csv | 17 ++ .../softroce-vm-paired-environment.json | 20 ++ .../results/softroce-vm-paired-samples.csv | 91 +++++++ .../results/softroce-vm-paired-summary.csv | 19 ++ .../results/softroce-vm-test-manifest.json | 117 +++++++++ .../results/softroce-vm-test-receipts.txt | 152 +++++++++++ .../results/softroce-vm-topology.json | 99 ++++++++ .../benchmarks/softroce_corrupt_restore.py | 96 +++++++ kv-service/configs/server-softroce-vm.toml | 47 ++++ kv-service/deploy/softroce-vm/build-verbs.sh | 11 + kv-service/deploy/softroce-vm/deploy-verbs.sh | 33 +++ .../softroce-vm/install-guest-prereqs.sh | 10 + kv-service/deploy/softroce-vm/prepare-pair.sh | 60 +++++ .../deploy/softroce-vm/setup-guest-paths.sh | 32 +++ .../deploy/softroce-vm/setup-host-bridges.sh | 27 ++ .../deploy/softroce-vm/ssh_config.example | 20 ++ .../deploy/softroce-vm/start-guest-server.sh | 33 +++ kv-service/deploy/softroce-vm/start-pair.sh | 27 ++ .../deploy/softroce-vm/stop-guest-server.sh | 8 + kv-service/deploy/softroce-vm/stop-pair.sh | 10 + .../softroce-vm/teardown-guest-paths.sh | 15 ++ .../softroce-vm/teardown-host-bridges.sh | 13 + 31 files changed, 1598 insertions(+) create mode 100644 kv-service/benchmarks/analyze_rail_verbs.py create mode 100644 kv-service/benchmarks/results/softroce-vm-analysis.json create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-256-batches.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-256-environment.json create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-256-samples.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-256-summary.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-64-128-batches.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-64-128-environment.json create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-64-128-samples.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-concurrent-64-128-summary.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-paired-environment.json create mode 100644 kv-service/benchmarks/results/softroce-vm-paired-samples.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-paired-summary.csv create mode 100644 kv-service/benchmarks/results/softroce-vm-test-manifest.json create mode 100644 kv-service/benchmarks/results/softroce-vm-test-receipts.txt create mode 100644 kv-service/benchmarks/results/softroce-vm-topology.json create mode 100644 kv-service/benchmarks/softroce_corrupt_restore.py create mode 100644 kv-service/configs/server-softroce-vm.toml create mode 100755 kv-service/deploy/softroce-vm/build-verbs.sh create mode 100755 kv-service/deploy/softroce-vm/deploy-verbs.sh create mode 100755 kv-service/deploy/softroce-vm/install-guest-prereqs.sh create mode 100755 kv-service/deploy/softroce-vm/prepare-pair.sh create mode 100755 kv-service/deploy/softroce-vm/setup-guest-paths.sh create mode 100755 kv-service/deploy/softroce-vm/setup-host-bridges.sh create mode 100644 kv-service/deploy/softroce-vm/ssh_config.example create mode 100755 kv-service/deploy/softroce-vm/start-guest-server.sh create mode 100755 kv-service/deploy/softroce-vm/start-pair.sh create mode 100755 kv-service/deploy/softroce-vm/stop-guest-server.sh create mode 100755 kv-service/deploy/softroce-vm/stop-pair.sh create mode 100755 kv-service/deploy/softroce-vm/teardown-guest-paths.sh create mode 100755 kv-service/deploy/softroce-vm/teardown-host-bridges.sh diff --git a/kv-service/benchmarks/analyze_rail_verbs.py b/kv-service/benchmarks/analyze_rail_verbs.py new file mode 100644 index 0000000..a4bf4f6 --- /dev/null +++ b/kv-service/benchmarks/analyze_rail_verbs.py @@ -0,0 +1,125 @@ +from __future__ import annotations + +# Summarize checked-in real-Verbs rail samples without inferential claims. + +import argparse +import csv +import json +import math +import statistics +from pathlib import Path + + +def read_csv(path: Path) -> list[dict[str, str]]: + with path.open(newline="") as handle: + return list(csv.DictReader(handle)) + + +def percentile_nearest_rank(values: list[int], percent: float) -> int: + ordered = sorted(values) + return ordered[max(0, math.ceil(percent * len(ordered)) - 1)] + + +def summarize_pair(results: Path, prefix: str) -> list[dict[str, object]]: + runs = read_csv(results / f"{prefix}-summary.csv") + samples = read_csv(results / f"{prefix}-samples.csv") + paired = [] + sizes = sorted({int(run["size_mib"]) for run in runs}) + concurrent = "concurrency" in runs[0] + for size in sizes: + levels = ( + sorted({int(run["concurrency"]) for run in runs if int(run["size_mib"]) == size}) + if concurrent + else [1] + ) + for concurrency in levels: + results_by_rail = {} + for rails in (1, 2): + selected = [ + run + for run in runs + if int(run["size_mib"]) == size + and int(run["rails"]) == rails + and int(run.get("concurrency", 1)) == concurrency + ] + sampled = [ + int(row["latency_us"]) + for row in samples + if int(row["size_mib"]) == size + and int(row["rails"]) == rails + and int(row.get("concurrency", 1)) == concurrency + ] + if not selected or not sampled: + raise ValueError(f"missing size={size} concurrency={concurrency} rail={rails}") + hashes = {run["xxh3"] for run in selected} + if len(hashes) != 1: + raise ValueError("object hash changed across runs") + if rails == 2 and not all( + int(run["rail0_bytes"]) > 0 and int(run["rail1_bytes"]) > 0 for run in selected + ): + raise ValueError("one rail transferred no payload") + median_us = statistics.median( + int(run["median_request_us" if concurrent else "median_us"]) for run in selected + ) + cpu_per_read_ms = statistics.median( + (int(run["cpu_user_us"]) + int(run["cpu_system_us"])) + / (int(run["batches" if concurrent else "iterations"]) * concurrency) + / 1000 + for run in selected + ) + throughput = ( + statistics.median(float(run["aggregate_gib_per_s"]) for run in selected) + if concurrent + else (size / 1024) / (median_us / 1_000_000) + ) + results_by_rail[rails] = dict( + rails=rails, + median_request_ms=median_us / 1000, + p95_request_ms=percentile_nearest_rank(sampled, 0.95) / 1000, + aggregate_gib_per_s=throughput, + cpu_ms_per_read=cpu_per_read_ms, + peak_rss_mib=statistics.median( + int(run["peak_rss_kb"]) / 1024 for run in selected + ), + object_xxh3=next(iter(hashes)), + runs=len(selected), + raw_request_samples=len(sampled), + ) + if results_by_rail[1]["object_xxh3"] != results_by_rail[2]["object_xxh3"]: + raise ValueError("single- and dual-rail payload hashes differ") + speedup = ( + results_by_rail[2]["aggregate_gib_per_s"] + / results_by_rail[1]["aggregate_gib_per_s"] + ) + paired.append( + dict( + size_mib=size, + concurrency=concurrency, + single=results_by_rail[1], + dual=results_by_rail[2], + aggregate_throughput_ratio=speedup, + two_rail_scaling_efficiency=speedup / 2, + ) + ) + return paired + + +def main() -> None: + parser = argparse.ArgumentParser(description="Summarize real-Verbs rail samples") + parser.add_argument("--results", type=Path, required=True) + parser.add_argument("--prefix", action="append", required=True) + parser.add_argument("--output", type=Path) + args = parser.parse_args() + analysis = { + "interpretation": "Descriptive paired measurements; no statistical confidence interval or HCA offload claim", + "groups": {prefix: summarize_pair(args.results, prefix) for prefix in args.prefix}, + } + rendered = json.dumps(analysis, indent=2) + "\n" + if args.output: + args.output.write_text(rendered) + else: + print(rendered, end="") + + +if __name__ == "__main__": + main() diff --git a/kv-service/benchmarks/results/softroce-vm-analysis.json b/kv-service/benchmarks/results/softroce-vm-analysis.json new file mode 100644 index 0000000..fc214a1 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-analysis.json @@ -0,0 +1,235 @@ +{ + "interpretation": "Descriptive paired measurements; no statistical confidence interval or HCA offload claim", + "groups": { + "softroce-vm-paired": [ + { + "size_mib": 64, + "concurrency": 1, + "single": { + "rails": 1, + "median_request_ms": 352.824, + "p95_request_ms": 394.774, + "aggregate_gib_per_s": 0.17714214452531574, + "cpu_ms_per_read": 123.132, + "peak_rss_mib": 198.2734375, + "object_xxh3": "a0a4cbfa5cad46af", + "runs": 3, + "raw_request_samples": 15 + }, + "dual": { + "rails": 2, + "median_request_ms": 329.447, + "p95_request_ms": 346.031, + "aggregate_gib_per_s": 0.18971185046456637, + "cpu_ms_per_read": 136.51239999999999, + "peak_rss_mib": 206.359375, + "object_xxh3": "a0a4cbfa5cad46af", + "runs": 3, + "raw_request_samples": 15 + }, + "aggregate_throughput_ratio": 1.0709583028529626, + "two_rail_scaling_efficiency": 0.5354791514264813 + }, + { + "size_mib": 128, + "concurrency": 1, + "single": { + "rails": 1, + "median_request_ms": 703.299, + "p95_request_ms": 748.001, + "aggregate_gib_per_s": 0.17773379458807706, + "cpu_ms_per_read": 235.071, + "peak_rss_mib": 390.31640625, + "object_xxh3": "6c7a2ea2bc98454c", + "runs": 3, + "raw_request_samples": 15 + }, + "dual": { + "rails": 2, + "median_request_ms": 647.801, + "p95_request_ms": 663.332, + "aggregate_gib_per_s": 0.19296049249692424, + "cpu_ms_per_read": 250.19920000000002, + "peak_rss_mib": 398.16015625, + "object_xxh3": "6c7a2ea2bc98454c", + "runs": 3, + "raw_request_samples": 15 + }, + "aggregate_throughput_ratio": 1.0856713713007544, + "two_rail_scaling_efficiency": 0.5428356856503772 + }, + { + "size_mib": 256, + "concurrency": 1, + "single": { + "rails": 1, + "median_request_ms": 1424.505, + "p95_request_ms": 1598.908, + "aggregate_gib_per_s": 0.17549955949610568, + "cpu_ms_per_read": 440.4614, + "peak_rss_mib": 774.2421875, + "object_xxh3": "fe51c755b1336171", + "runs": 3, + "raw_request_samples": 15 + }, + "dual": { + "rails": 2, + "median_request_ms": 1279.419, + "p95_request_ms": 1389.539, + "aggregate_gib_per_s": 0.19540119382313376, + "cpu_ms_per_read": 463.818, + "peak_rss_mib": 782.30078125, + "object_xxh3": "fe51c755b1336171", + "runs": 3, + "raw_request_samples": 15 + }, + "aggregate_throughput_ratio": 1.1133999104280925, + "two_rail_scaling_efficiency": 0.5566999552140462 + } + ], + "softroce-vm-concurrent-64-128": [ + { + "size_mib": 64, + "concurrency": 2, + "single": { + "rails": 1, + "median_request_ms": 581.23, + "p95_request_ms": 599.33, + "aggregate_gib_per_s": 0.207, + "cpu_ms_per_read": 140.90866666666665, + "peak_rss_mib": 454.52734375, + "object_xxh3": "a0a4cbfa5cad46af", + "runs": 2, + "raw_request_samples": 12 + }, + "dual": { + "rails": 2, + "median_request_ms": 522.294, + "p95_request_ms": 536.04, + "aggregate_gib_per_s": 0.2265, + "cpu_ms_per_read": 150.62341666666669, + "peak_rss_mib": 470.71484375, + "object_xxh3": "a0a4cbfa5cad46af", + "runs": 2, + "raw_request_samples": 12 + }, + "aggregate_throughput_ratio": 1.0942028985507248, + "two_rail_scaling_efficiency": 0.5471014492753624 + }, + { + "size_mib": 64, + "concurrency": 4, + "single": { + "rails": 1, + "median_request_ms": 934.8985, + "p95_request_ms": 987.519, + "aggregate_gib_per_s": 0.2525, + "cpu_ms_per_read": 148.35825, + "peak_rss_mib": 837.662109375, + "object_xxh3": "a0a4cbfa5cad46af", + "runs": 2, + "raw_request_samples": 24 + }, + "dual": { + "rails": 2, + "median_request_ms": 912.668, + "p95_request_ms": 959.521, + "aggregate_gib_per_s": 0.257, + "cpu_ms_per_read": 153.86858333333333, + "peak_rss_mib": 815.591796875, + "object_xxh3": "a0a4cbfa5cad46af", + "runs": 2, + "raw_request_samples": 24 + }, + "aggregate_throughput_ratio": 1.0178217821782178, + "two_rail_scaling_efficiency": 0.5089108910891089 + }, + { + "size_mib": 128, + "concurrency": 2, + "single": { + "rails": 1, + "median_request_ms": 1097.5775, + "p95_request_ms": 1157.407, + "aggregate_gib_per_s": 0.2135, + "cpu_ms_per_read": 260.68516666666665, + "peak_rss_mib": 789.76953125, + "object_xxh3": "6c7a2ea2bc98454c", + "runs": 2, + "raw_request_samples": 12 + }, + "dual": { + "rails": 2, + "median_request_ms": 1008.9295, + "p95_request_ms": 1042.112, + "aggregate_gib_per_s": 0.238, + "cpu_ms_per_read": 279.73075, + "peak_rss_mib": 918.78515625, + "object_xxh3": "6c7a2ea2bc98454c", + "runs": 2, + "raw_request_samples": 12 + }, + "aggregate_throughput_ratio": 1.1147540983606556, + "two_rail_scaling_efficiency": 0.5573770491803278 + }, + { + "size_mib": 128, + "concurrency": 4, + "single": { + "rails": 1, + "median_request_ms": 1902.2805, + "p95_request_ms": 1965.519, + "aggregate_gib_per_s": 0.2485, + "cpu_ms_per_read": 272.177, + "peak_rss_mib": 1371.53125, + "object_xxh3": "6c7a2ea2bc98454c", + "runs": 2, + "raw_request_samples": 24 + }, + "dual": { + "rails": 2, + "median_request_ms": 1764.8325, + "p95_request_ms": 1847.11, + "aggregate_gib_per_s": 0.268, + "cpu_ms_per_read": 283.3401666666667, + "peak_rss_mib": 1555.58984375, + "object_xxh3": "6c7a2ea2bc98454c", + "runs": 2, + "raw_request_samples": 24 + }, + "aggregate_throughput_ratio": 1.0784708249496984, + "two_rail_scaling_efficiency": 0.5392354124748492 + } + ], + "softroce-vm-concurrent-256": [ + { + "size_mib": 256, + "concurrency": 2, + "single": { + "rails": 1, + "median_request_ms": 2315.1275, + "p95_request_ms": 2487.62, + "aggregate_gib_per_s": 0.20750000000000002, + "cpu_ms_per_read": 507.721, + "peak_rss_mib": 1791.810546875, + "object_xxh3": "fe51c755b1336171", + "runs": 2, + "raw_request_samples": 12 + }, + "dual": { + "rails": 2, + "median_request_ms": 1983.32, + "p95_request_ms": 2079.975, + "aggregate_gib_per_s": 0.2395, + "cpu_ms_per_read": 525.6972499999999, + "peak_rss_mib": 1814.5546875, + "object_xxh3": "fe51c755b1336171", + "runs": 2, + "raw_request_samples": 12 + }, + "aggregate_throughput_ratio": 1.1542168674698794, + "two_rail_scaling_efficiency": 0.5771084337349397 + } + ] + } +} diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-256-batches.csv b/kv-service/benchmarks/results/softroce-vm-concurrent-256-batches.csv new file mode 100644 index 0000000..e8942bf --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-256-batches.csv @@ -0,0 +1,13 @@ +trial,size_mib,rails,concurrency,batch,wall_us +1,256,1,2,1,2569287 +1,256,1,2,2,2449074 +1,256,1,2,3,2379167 +1,256,2,2,1,2096616 +1,256,2,2,2,2057070 +1,256,2,2,3,2128493 +2,256,2,2,1,2023836 +2,256,2,2,2,2122626 +2,256,2,2,3,2115578 +2,256,1,2,1,2408893 +2,256,1,2,2,2322398 +2,256,1,2,3,2328700 diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-256-environment.json b/kv-service/benchmarks/results/softroce-vm-concurrent-256-environment.json new file mode 100644 index 0000000..d61ad25 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-256-environment.json @@ -0,0 +1,21 @@ +{ + "environment": "soft-roce", + "source_commit": "cd529b5", + "platform": "Linux-5.15.0-194-generic-x86_64-with-glibc2.35", + "machine": "x86_64", + "notes": "Two KVM guests, two virtual RXE rails, 4 vCPU and 4 GiB each; RLIMIT_MEMLOCK raised to unlimited inside the isolated client guest for a fair concurrent matrix", + "objects_mib": [ + 256 + ], + "concurrency": [ + 2 + ], + "trials": 2, + "batches_per_trial": 3, + "warmup_reads": 1, + "coordinator": "http://10.31.0.2:55151", + "rails": [ + "r0,rxe_c0,10.31.0.2:55153,10.31.0.2:55153,1,1", + "r1,rxe_c1,10.31.0.2:55153,10.32.0.2:55154,1,1" + ] +} diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-256-samples.csv b/kv-service/benchmarks/results/softroce-vm-concurrent-256-samples.csv new file mode 100644 index 0000000..63d7e8f --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-256-samples.csv @@ -0,0 +1,25 @@ +trial,size_mib,rails,concurrency,batch,worker,bytes,latency_us,xxh3 +1,256,1,2,1,0,268435456,2487620,fe51c755b1336171 +1,256,1,2,1,1,268435456,2351607,fe51c755b1336171 +1,256,1,2,2,0,268435456,2370370,fe51c755b1336171 +1,256,1,2,2,1,268435456,2285753,fe51c755b1336171 +1,256,1,2,3,0,268435456,2298482,fe51c755b1336171 +1,256,1,2,3,1,268435456,2210071,fe51c755b1336171 +1,256,2,2,1,0,268435456,2013569,fe51c755b1336171 +1,256,2,2,1,1,268435456,1929027,fe51c755b1336171 +1,256,2,2,2,0,268435456,1972182,fe51c755b1336171 +1,256,2,2,2,1,268435456,1885959,fe51c755b1336171 +1,256,2,2,3,0,268435456,2046871,fe51c755b1336171 +1,256,2,2,3,1,268435456,1962123,fe51c755b1336171 +2,256,2,2,1,0,268435456,1894955,fe51c755b1336171 +2,256,2,2,1,1,268435456,1980073,fe51c755b1336171 +2,256,2,2,2,0,268435456,1994458,fe51c755b1336171 +2,256,2,2,2,1,268435456,2079975,fe51c755b1336171 +2,256,2,2,3,0,268435456,2034204,fe51c755b1336171 +2,256,2,2,3,1,268435456,1950234,fe51c755b1336171 +2,256,1,2,1,0,268435456,2327837,fe51c755b1336171 +2,256,1,2,1,1,268435456,2241328,fe51c755b1336171 +2,256,1,2,2,0,268435456,2195260,fe51c755b1336171 +2,256,1,2,2,1,268435456,2278648,fe51c755b1336171 +2,256,1,2,3,0,268435456,2204936,fe51c755b1336171 +2,256,1,2,3,1,268435456,2288398,fe51c755b1336171 diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-256-summary.csv b/kv-service/benchmarks/results/softroce-vm-concurrent-256-summary.csv new file mode 100644 index 0000000..b2de60c --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-256-summary.csv @@ -0,0 +1,5 @@ +trial,size_mib,rails,concurrency,batches,median_request_us,median_batch_us,aggregate_gib_per_s,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes,xxh3 +1,256,1,2,3,2351607,2449074,0.203,1455292,1628432,1827932,1610612736,0,fe51c755b1336171 +1,256,2,2,3,1972182,2096616,0.239,1453505,1693725,1857996,805306368,805306368,fe51c755b1336171 +2,256,2,2,3,1994458,2115578,0.24,1505317,1655820,1858212,805306368,805306368,fe51c755b1336171 +2,256,1,2,3,2278648,2328700,0.212,1438045,1570883,1841696,1610612736,0,fe51c755b1336171 diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-batches.csv b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-batches.csv new file mode 100644 index 0000000..5e86955 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-batches.csv @@ -0,0 +1,49 @@ +trial,size_mib,rails,concurrency,batch,wall_us +1,64,1,2,1,615610 +1,64,1,2,2,595422 +1,64,1,2,3,611357 +1,64,2,2,1,546773 +1,64,2,2,2,547463 +1,64,2,2,3,553001 +1,64,1,4,1,1019551 +1,64,1,4,2,1010455 +1,64,1,4,3,1011160 +1,64,2,4,1,947494 +1,64,2,4,2,975539 +1,64,2,4,3,957068 +1,128,1,2,1,1189924 +1,128,1,2,2,1119312 +1,128,1,2,3,1157775 +1,128,2,2,1,1042744 +1,128,2,2,2,1066373 +1,128,2,2,3,1055354 +1,128,1,4,1,2064187 +1,128,1,4,2,1956892 +1,128,1,4,3,1995798 +1,128,2,4,1,1883886 +1,128,2,4,2,1811513 +1,128,2,4,3,1888693 +2,64,2,2,1,538094 +2,64,2,2,2,561776 +2,64,2,2,3,563873 +2,64,1,2,1,601795 +2,64,1,2,2,594177 +2,64,1,2,3,610473 +2,64,2,4,1,971862 +2,64,2,4,2,972943 +2,64,2,4,3,1003945 +2,64,1,4,1,1010630 +2,64,1,4,2,928326 +2,64,1,4,3,966798 +2,128,2,2,1,1044705 +2,128,2,2,2,1024141 +2,128,2,2,3,1063227 +2,128,1,2,1,1185936 +2,128,1,2,2,1201239 +2,128,1,2,3,1161484 +2,128,2,4,1,1885758 +2,128,2,4,2,1892976 +2,128,2,4,3,1844462 +2,128,1,4,1,2006665 +2,128,1,4,2,2003782 +2,128,1,4,3,2028660 diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-environment.json b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-environment.json new file mode 100644 index 0000000..a3d3028 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-environment.json @@ -0,0 +1,23 @@ +{ + "environment": "soft-roce", + "source_commit": "cd529b5", + "platform": "Linux-5.15.0-194-generic-x86_64-with-glibc2.35", + "machine": "x86_64", + "notes": "Two KVM guests, two virtual RXE rails, 4 vCPU and 4 GiB each; RLIMIT_MEMLOCK raised to unlimited inside the isolated client guest for a fair concurrent matrix", + "objects_mib": [ + 64, + 128 + ], + "concurrency": [ + 2, + 4 + ], + "trials": 2, + "batches_per_trial": 3, + "warmup_reads": 1, + "coordinator": "http://10.31.0.2:55151", + "rails": [ + "r0,rxe_c0,10.31.0.2:55153,10.31.0.2:55153,1,1", + "r1,rxe_c1,10.31.0.2:55153,10.32.0.2:55154,1,1" + ] +} diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-samples.csv b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-samples.csv new file mode 100644 index 0000000..c0190c6 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-samples.csv @@ -0,0 +1,145 @@ +trial,size_mib,rails,concurrency,batch,worker,bytes,latency_us,xxh3 +1,64,1,2,1,0,67108864,589473,a0a4cbfa5cad46af +1,64,1,2,1,1,67108864,566411,a0a4cbfa5cad46af +1,64,1,2,2,0,67108864,561469,a0a4cbfa5cad46af +1,64,1,2,2,1,67108864,584982,a0a4cbfa5cad46af +1,64,1,2,3,0,67108864,575696,a0a4cbfa5cad46af +1,64,1,2,3,1,67108864,599330,a0a4cbfa5cad46af +1,64,2,2,1,0,67108864,517549,a0a4cbfa5cad46af +1,64,2,2,1,1,67108864,496469,a0a4cbfa5cad46af +1,64,2,2,2,0,67108864,521243,a0a4cbfa5cad46af +1,64,2,2,2,1,67108864,496644,a0a4cbfa5cad46af +1,64,2,2,3,0,67108864,526186,a0a4cbfa5cad46af +1,64,2,2,3,1,67108864,502366,a0a4cbfa5cad46af +1,64,1,4,1,0,67108864,937395,a0a4cbfa5cad46af +1,64,1,4,1,1,67108864,984711,a0a4cbfa5cad46af +1,64,1,4,1,2,67108864,987519,a0a4cbfa5cad46af +1,64,1,4,1,3,67108864,963491,a0a4cbfa5cad46af +1,64,1,4,2,0,67108864,901582,a0a4cbfa5cad46af +1,64,1,4,2,1,67108864,928737,a0a4cbfa5cad46af +1,64,1,4,2,2,67108864,983335,a0a4cbfa5cad46af +1,64,1,4,2,3,67108864,957952,a0a4cbfa5cad46af +1,64,1,4,3,0,67108864,967809,a0a4cbfa5cad46af +1,64,1,4,3,1,67108864,921039,a0a4cbfa5cad46af +1,64,1,4,3,2,67108864,965358,a0a4cbfa5cad46af +1,64,1,4,3,3,67108864,943612,a0a4cbfa5cad46af +1,64,2,4,1,0,67108864,900144,a0a4cbfa5cad46af +1,64,2,4,1,1,67108864,875764,a0a4cbfa5cad46af +1,64,2,4,1,2,67108864,835630,a0a4cbfa5cad46af +1,64,2,4,1,3,67108864,808354,a0a4cbfa5cad46af +1,64,2,4,2,0,67108864,928190,a0a4cbfa5cad46af +1,64,2,4,2,1,67108864,903404,a0a4cbfa5cad46af +1,64,2,4,2,2,67108864,954713,a0a4cbfa5cad46af +1,64,2,4,2,3,67108864,838185,a0a4cbfa5cad46af +1,64,2,4,3,0,67108864,871546,a0a4cbfa5cad46af +1,64,2,4,3,1,67108864,924361,a0a4cbfa5cad46af +1,64,2,4,3,2,67108864,921701,a0a4cbfa5cad46af +1,64,2,4,3,3,67108864,900159,a0a4cbfa5cad46af +1,128,1,2,1,0,134217728,1145747,6c7a2ea2bc98454c +1,128,1,2,1,1,134217728,1053300,6c7a2ea2bc98454c +1,128,1,2,2,0,134217728,1073565,6c7a2ea2bc98454c +1,128,1,2,2,1,134217728,974077,6c7a2ea2bc98454c +1,128,1,2,3,0,134217728,1116763,6c7a2ea2bc98454c +1,128,1,2,3,1,134217728,996645,6c7a2ea2bc98454c +1,128,2,2,1,0,134217728,972532,6c7a2ea2bc98454c +1,128,2,2,1,1,134217728,1018873,6c7a2ea2bc98454c +1,128,2,2,2,0,134217728,994805,6c7a2ea2bc98454c +1,128,2,2,2,1,134217728,1042112,6c7a2ea2bc98454c +1,128,2,2,3,0,134217728,985053,6c7a2ea2bc98454c +1,128,2,2,3,1,134217728,1030946,6c7a2ea2bc98454c +1,128,1,4,1,0,134217728,1817189,6c7a2ea2bc98454c +1,128,1,4,1,1,134217728,2000491,6c7a2ea2bc98454c +1,128,1,4,1,2,134217728,1955195,6c7a2ea2bc98454c +1,128,1,4,1,3,134217728,1864747,6c7a2ea2bc98454c +1,128,1,4,2,0,134217728,1673567,6c7a2ea2bc98454c +1,128,1,4,2,1,134217728,1889414,6c7a2ea2bc98454c +1,128,1,4,2,2,134217728,1844832,6c7a2ea2bc98454c +1,128,1,4,2,3,134217728,1937197,6c7a2ea2bc98454c +1,128,1,4,3,0,134217728,1827638,6c7a2ea2bc98454c +1,128,1,4,3,1,134217728,1940265,6c7a2ea2bc98454c +1,128,1,4,3,2,134217728,1695609,6c7a2ea2bc98454c +1,128,1,4,3,3,134217728,1896112,6c7a2ea2bc98454c +1,128,2,4,1,0,134217728,1731021,6c7a2ea2bc98454c +1,128,2,4,1,1,134217728,1818780,6c7a2ea2bc98454c +1,128,2,4,1,2,134217728,1684422,6c7a2ea2bc98454c +1,128,2,4,1,3,134217728,1775973,6c7a2ea2bc98454c +1,128,2,4,2,0,134217728,1706102,6c7a2ea2bc98454c +1,128,2,4,2,1,134217728,1751885,6c7a2ea2bc98454c +1,128,2,4,2,2,134217728,1757298,6c7a2ea2bc98454c +1,128,2,4,2,3,134217728,1660045,6c7a2ea2bc98454c +1,128,2,4,3,0,134217728,1805180,6c7a2ea2bc98454c +1,128,2,4,3,1,134217728,1715334,6c7a2ea2bc98454c +1,128,2,4,3,2,134217728,1762376,6c7a2ea2bc98454c +1,128,2,4,3,3,134217728,1616810,6c7a2ea2bc98454c +2,64,2,2,1,0,67108864,502024,a0a4cbfa5cad46af +2,64,2,2,1,1,67108864,527039,a0a4cbfa5cad46af +2,64,2,2,2,0,67108864,533335,a0a4cbfa5cad46af +2,64,2,2,2,1,67108864,509643,a0a4cbfa5cad46af +2,64,2,2,3,0,67108864,536040,a0a4cbfa5cad46af +2,64,2,2,3,1,67108864,515018,a0a4cbfa5cad46af +2,64,1,2,1,0,67108864,577478,a0a4cbfa5cad46af +2,64,1,2,1,1,67108864,553261,a0a4cbfa5cad46af +2,64,1,2,2,0,67108864,559305,a0a4cbfa5cad46af +2,64,1,2,2,1,67108864,584740,a0a4cbfa5cad46af +2,64,1,2,3,0,67108864,582224,a0a4cbfa5cad46af +2,64,1,2,3,1,67108864,559301,a0a4cbfa5cad46af +2,64,2,4,1,0,67108864,859220,a0a4cbfa5cad46af +2,64,2,4,1,1,67108864,934980,a0a4cbfa5cad46af +2,64,2,4,1,2,67108864,913365,a0a4cbfa5cad46af +2,64,2,4,1,3,67108864,890051,a0a4cbfa5cad46af +2,64,2,4,2,0,67108864,878721,a0a4cbfa5cad46af +2,64,2,4,2,1,67108864,925177,a0a4cbfa5cad46af +2,64,2,4,2,2,67108864,950761,a0a4cbfa5cad46af +2,64,2,4,2,3,67108864,902095,a0a4cbfa5cad46af +2,64,2,4,3,0,67108864,936306,a0a4cbfa5cad46af +2,64,2,4,3,1,67108864,893172,a0a4cbfa5cad46af +2,64,2,4,3,2,67108864,980890,a0a4cbfa5cad46af +2,64,2,4,3,3,67108864,959521,a0a4cbfa5cad46af +2,64,1,4,1,0,67108864,964845,a0a4cbfa5cad46af +2,64,1,4,1,1,67108864,908543,a0a4cbfa5cad46af +2,64,1,4,1,2,67108864,940054,a0a4cbfa5cad46af +2,64,1,4,1,3,67108864,988118,a0a4cbfa5cad46af +2,64,1,4,2,0,67108864,882260,a0a4cbfa5cad46af +2,64,1,4,2,1,67108864,837238,a0a4cbfa5cad46af +2,64,1,4,2,2,67108864,879968,a0a4cbfa5cad46af +2,64,1,4,2,3,67108864,858689,a0a4cbfa5cad46af +2,64,1,4,3,0,67108864,929016,a0a4cbfa5cad46af +2,64,1,4,3,1,67108864,784029,a0a4cbfa5cad46af +2,64,1,4,3,2,67108864,906306,a0a4cbfa5cad46af +2,64,1,4,3,3,67108864,861196,a0a4cbfa5cad46af +2,128,2,2,1,0,134217728,978161,6c7a2ea2bc98454c +2,128,2,2,1,1,134217728,1021389,6c7a2ea2bc98454c +2,128,2,2,2,0,134217728,952177,6c7a2ea2bc98454c +2,128,2,2,2,1,134217728,998986,6c7a2ea2bc98454c +2,128,2,2,3,0,134217728,990150,6c7a2ea2bc98454c +2,128,2,2,3,1,134217728,1037344,6c7a2ea2bc98454c +2,128,1,2,1,0,134217728,1143835,6c7a2ea2bc98454c +2,128,1,2,1,1,134217728,1060955,6c7a2ea2bc98454c +2,128,1,2,2,0,134217728,1157407,6c7a2ea2bc98454c +2,128,1,2,2,1,134217728,1049618,6c7a2ea2bc98454c +2,128,1,2,3,0,134217728,1121590,6c7a2ea2bc98454c +2,128,1,2,3,1,134217728,1047067,6c7a2ea2bc98454c +2,128,2,4,1,0,134217728,1800598,6c7a2ea2bc98454c +2,128,2,4,1,1,134217728,1667007,6c7a2ea2bc98454c +2,128,2,4,1,2,134217728,1847110,6c7a2ea2bc98454c +2,128,2,4,1,3,134217728,1755109,6c7a2ea2bc98454c +2,128,2,4,2,0,134217728,1807005,6c7a2ea2bc98454c +2,128,2,4,2,1,134217728,1758174,6c7a2ea2bc98454c +2,128,2,4,2,2,134217728,1850573,6c7a2ea2bc98454c +2,128,2,4,2,3,134217728,1713594,6c7a2ea2bc98454c +2,128,2,4,3,0,134217728,1735551,6c7a2ea2bc98454c +2,128,2,4,3,1,134217728,1689324,6c7a2ea2bc98454c +2,128,2,4,3,2,134217728,1777780,6c7a2ea2bc98454c +2,128,2,4,3,3,134217728,1821223,6c7a2ea2bc98454c +2,128,1,4,1,0,134217728,1926933,6c7a2ea2bc98454c +2,128,1,4,1,1,134217728,1882021,6c7a2ea2bc98454c +2,128,1,4,1,2,134217728,1931719,6c7a2ea2bc98454c +2,128,1,4,1,3,134217728,1769127,6c7a2ea2bc98454c +2,128,1,4,2,0,134217728,1857819,6c7a2ea2bc98454c +2,128,1,4,2,1,134217728,1915147,6c7a2ea2bc98454c +2,128,1,4,2,2,134217728,1960477,6c7a2ea2bc98454c +2,128,1,4,2,3,134217728,1798472,6c7a2ea2bc98454c +2,128,1,4,3,0,134217728,1735327,6c7a2ea2bc98454c +2,128,1,4,3,1,134217728,1965519,6c7a2ea2bc98454c +2,128,1,4,3,2,134217728,1836042,6c7a2ea2bc98454c +2,128,1,4,3,3,134217728,1918094,6c7a2ea2bc98454c diff --git a/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-summary.csv b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-summary.csv new file mode 100644 index 0000000..fd40385 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-concurrent-64-128-summary.csv @@ -0,0 +1,17 @@ +trial,size_mib,rails,concurrency,batches,median_request_us,median_batch_us,aggregate_gib_per_s,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes,xxh3 +1,64,1,2,3,584982,611357,0.206,440721,417074,465624,402653184,0,a0a4cbfa5cad46af +1,64,2,2,3,517549,547463,0.228,435376,457823,482064,201326592,201326592,a0a4cbfa5cad46af +1,64,1,4,3,963491,1011160,0.247,835863,939721,858216,805306368,0,a0a4cbfa5cad46af +1,64,2,4,3,900159,957068,0.26,796104,999099,873036,402653184,402653184,a0a4cbfa5cad46af +1,128,1,2,3,1073565,1157775,0.216,778757,798235,793308,805306368,0,6c7a2ea2bc98454c +1,128,2,2,3,1018873,1055354,0.237,815531,877221,940816,402653184,402653184,6c7a2ea2bc98454c +1,128,1,4,3,1889414,1995798,0.249,1517768,1782414,1374280,1610612736,0,6c7a2ea2bc98454c +1,128,2,4,3,1751885,1883886,0.269,1551378,1910349,1545312,805306368,805306368,6c7a2ea2bc98454c +2,64,2,2,3,527039,561776,0.225,449637,464645,481960,201326592,201326592,a0a4cbfa5cad46af +2,64,1,2,3,577478,601795,0.208,380840,452269,465248,402653184,0,a0a4cbfa5cad46af +2,64,2,4,3,925177,972943,0.254,845043,1052600,797296,402653184,402653184,a0a4cbfa5cad46af +2,64,1,4,3,906306,966798,0.258,806998,978016,857316,805306368,0,a0a4cbfa5cad46af +2,128,2,2,3,998986,1044705,0.239,828891,835126,940856,402653184,402653184,6c7a2ea2bc98454c +2,128,1,2,3,1121590,1185936,0.211,734777,816453,824140,805306368,0,6c7a2ea2bc98454c +2,128,2,4,3,1777780,1885758,0.267,1519564,1818873,1640536,805306368,805306368,6c7a2ea2bc98454c +2,128,1,4,3,1915147,2006665,0.248,1503054,1729012,1434616,1610612736,0,6c7a2ea2bc98454c diff --git a/kv-service/benchmarks/results/softroce-vm-paired-environment.json b/kv-service/benchmarks/results/softroce-vm-paired-environment.json new file mode 100644 index 0000000..dc25659 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-paired-environment.json @@ -0,0 +1,20 @@ +{ + "environment": "soft-roce", + "source_commit": "52b933d", + "platform": "Linux-5.15.0-194-generic-x86_64-with-glibc2.35", + "machine": "x86_64", + "notes": "Two Ubuntu 22.04 KVM guests, two isolated Linux bridges and virtio Ethernet links, four RXE devices, 4 vCPU and 4 GiB each, two stripe directories on one guest virtual disk", + "objects_mib": [ + 64, + 128, + 256 + ], + "trials": 3, + "iterations_per_trial": 5, + "warmup_reads": 1, + "coordinator": "http://10.31.0.2:55151", + "rails": [ + "r0,rxe_c0,10.31.0.2:55153,10.31.0.2:55153,1,1", + "r1,rxe_c1,10.31.0.2:55153,10.32.0.2:55154,1,1" + ] +} diff --git a/kv-service/benchmarks/results/softroce-vm-paired-samples.csv b/kv-service/benchmarks/results/softroce-vm-paired-samples.csv new file mode 100644 index 0000000..45ed8f3 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-paired-samples.csv @@ -0,0 +1,91 @@ +trial,size_mib,rails,iteration,bytes,latency_us,gib_per_s,xxh3 +1,64,1,1,67108864,373991,0.167,a0a4cbfa5cad46af +1,64,1,2,67108864,352824,0.177,a0a4cbfa5cad46af +1,64,1,3,67108864,351315,0.178,a0a4cbfa5cad46af +1,64,1,4,67108864,361752,0.173,a0a4cbfa5cad46af +1,64,1,5,67108864,352155,0.177,a0a4cbfa5cad46af +1,64,2,1,67108864,337882,0.185,a0a4cbfa5cad46af +1,64,2,2,67108864,326523,0.191,a0a4cbfa5cad46af +1,64,2,3,67108864,333263,0.188,a0a4cbfa5cad46af +1,64,2,4,67108864,329447,0.19,a0a4cbfa5cad46af +1,64,2,5,67108864,324548,0.193,a0a4cbfa5cad46af +1,128,1,1,134217728,721789,0.173,6c7a2ea2bc98454c +1,128,1,2,134217728,695234,0.18,6c7a2ea2bc98454c +1,128,1,3,134217728,716867,0.174,6c7a2ea2bc98454c +1,128,1,4,134217728,703299,0.178,6c7a2ea2bc98454c +1,128,1,5,134217728,681577,0.183,6c7a2ea2bc98454c +1,128,2,1,134217728,651498,0.192,6c7a2ea2bc98454c +1,128,2,2,134217728,641352,0.195,6c7a2ea2bc98454c +1,128,2,3,134217728,647801,0.193,6c7a2ea2bc98454c +1,128,2,4,134217728,644689,0.194,6c7a2ea2bc98454c +1,128,2,5,134217728,663332,0.188,6c7a2ea2bc98454c +1,256,1,1,268435456,1492634,0.167,fe51c755b1336171 +1,256,1,2,268435456,1432946,0.174,fe51c755b1336171 +1,256,1,3,268435456,1446049,0.173,fe51c755b1336171 +1,256,1,4,268435456,1426279,0.175,fe51c755b1336171 +1,256,1,5,268435456,1431284,0.175,fe51c755b1336171 +1,256,2,1,268435456,1243102,0.201,fe51c755b1336171 +1,256,2,2,268435456,1268669,0.197,fe51c755b1336171 +1,256,2,3,268435456,1206837,0.207,fe51c755b1336171 +1,256,2,4,268435456,1238127,0.202,fe51c755b1336171 +1,256,2,5,268435456,1235937,0.202,fe51c755b1336171 +2,64,2,1,67108864,335713,0.186,a0a4cbfa5cad46af +2,64,2,2,67108864,333849,0.187,a0a4cbfa5cad46af +2,64,2,3,67108864,325205,0.192,a0a4cbfa5cad46af +2,64,2,4,67108864,328238,0.19,a0a4cbfa5cad46af +2,64,2,5,67108864,322567,0.194,a0a4cbfa5cad46af +2,64,1,1,67108864,371380,0.168,a0a4cbfa5cad46af +2,64,1,2,67108864,357153,0.175,a0a4cbfa5cad46af +2,64,1,3,67108864,371129,0.168,a0a4cbfa5cad46af +2,64,1,4,67108864,381567,0.164,a0a4cbfa5cad46af +2,64,1,5,67108864,394774,0.158,a0a4cbfa5cad46af +2,128,2,1,134217728,649725,0.192,6c7a2ea2bc98454c +2,128,2,2,134217728,637599,0.196,6c7a2ea2bc98454c +2,128,2,3,134217728,627598,0.199,6c7a2ea2bc98454c +2,128,2,4,134217728,657023,0.19,6c7a2ea2bc98454c +2,128,2,5,134217728,652498,0.192,6c7a2ea2bc98454c +2,128,1,1,134217728,737776,0.169,6c7a2ea2bc98454c +2,128,1,2,134217728,727485,0.172,6c7a2ea2bc98454c +2,128,1,3,134217728,725839,0.172,6c7a2ea2bc98454c +2,128,1,4,134217728,747635,0.167,6c7a2ea2bc98454c +2,128,1,5,134217728,748001,0.167,6c7a2ea2bc98454c +2,256,2,1,268435456,1215398,0.206,fe51c755b1336171 +2,256,2,2,268435456,1272247,0.197,fe51c755b1336171 +2,256,2,3,268435456,1279419,0.195,fe51c755b1336171 +2,256,2,4,268435456,1283626,0.195,fe51c755b1336171 +2,256,2,5,268435456,1370607,0.182,fe51c755b1336171 +2,256,1,1,268435456,1476390,0.169,fe51c755b1336171 +2,256,1,2,268435456,1441234,0.173,fe51c755b1336171 +2,256,1,3,268435456,1392140,0.18,fe51c755b1336171 +2,256,1,4,268435456,1397353,0.179,fe51c755b1336171 +2,256,1,5,268435456,1411231,0.177,fe51c755b1336171 +3,64,1,1,67108864,347734,0.18,a0a4cbfa5cad46af +3,64,1,2,67108864,352691,0.177,a0a4cbfa5cad46af +3,64,1,3,67108864,356587,0.175,a0a4cbfa5cad46af +3,64,1,4,67108864,352093,0.178,a0a4cbfa5cad46af +3,64,1,5,67108864,357157,0.175,a0a4cbfa5cad46af +3,64,2,1,67108864,333433,0.187,a0a4cbfa5cad46af +3,64,2,2,67108864,327670,0.191,a0a4cbfa5cad46af +3,64,2,3,67108864,346031,0.181,a0a4cbfa5cad46af +3,64,2,4,67108864,330370,0.189,a0a4cbfa5cad46af +3,64,2,5,67108864,330358,0.189,a0a4cbfa5cad46af +3,128,1,1,134217728,693713,0.18,6c7a2ea2bc98454c +3,128,1,2,134217728,695636,0.18,6c7a2ea2bc98454c +3,128,1,3,134217728,674564,0.185,6c7a2ea2bc98454c +3,128,1,4,134217728,687005,0.182,6c7a2ea2bc98454c +3,128,1,5,134217728,682664,0.183,6c7a2ea2bc98454c +3,128,2,1,134217728,609910,0.205,6c7a2ea2bc98454c +3,128,2,2,134217728,635044,0.197,6c7a2ea2bc98454c +3,128,2,3,134217728,644442,0.194,6c7a2ea2bc98454c +3,128,2,4,134217728,614333,0.203,6c7a2ea2bc98454c +3,128,2,5,134217728,615369,0.203,6c7a2ea2bc98454c +3,256,1,1,268435456,1416214,0.177,fe51c755b1336171 +3,256,1,2,268435456,1446984,0.173,fe51c755b1336171 +3,256,1,3,268435456,1417506,0.176,fe51c755b1336171 +3,256,1,4,268435456,1424505,0.175,fe51c755b1336171 +3,256,1,5,268435456,1598908,0.156,fe51c755b1336171 +3,256,2,1,268435456,1202171,0.208,fe51c755b1336171 +3,256,2,2,268435456,1389539,0.18,fe51c755b1336171 +3,256,2,3,268435456,1343528,0.186,fe51c755b1336171 +3,256,2,4,268435456,1248144,0.2,fe51c755b1336171 +3,256,2,5,268435456,1309757,0.191,fe51c755b1336171 diff --git a/kv-service/benchmarks/results/softroce-vm-paired-summary.csv b/kv-service/benchmarks/results/softroce-vm-paired-summary.csv new file mode 100644 index 0000000..7822da9 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-paired-summary.csv @@ -0,0 +1,19 @@ +trial,size_mib,rails,iterations,median_us,cpu_user_us,cpu_system_us,peak_rss_kb,rail0_bytes,rail1_bytes,xxh3 +1,64,1,5,352824,271884,343776,203044,335544320,0,a0a4cbfa5cad46af +1,64,2,5,329447,285274,395693,211360,167772160,167772160,a0a4cbfa5cad46af +1,128,1,5,703299,523828,651527,399608,671088640,0,6c7a2ea2bc98454c +1,128,2,5,647801,550518,700478,407544,335544320,335544320,6c7a2ea2bc98454c +1,256,1,5,1432946,977265,1279736,792776,1342177280,0,fe51c755b1336171 +1,256,2,5,1238127,1043272,1275818,801036,671088640,671088640,fe51c755b1336171 +2,64,2,5,328238,304656,377906,211044,167772160,167772160,a0a4cbfa5cad46af +2,64,1,5,371380,335035,351144,202808,335544320,0,a0a4cbfa5cad46af +2,128,2,5,649725,562037,759990,407716,335544320,335544320,6c7a2ea2bc98454c +2,128,1,5,737776,715828,634447,399904,671088640,0,6c7a2ea2bc98454c +2,256,2,5,1279419,1128224,1421902,801076,671088640,671088640,fe51c755b1336171 +2,256,1,5,1411231,1038584,1126219,793100,1342177280,0,fe51c755b1336171 +3,64,1,5,352691,283027,295316,203032,335544320,0,a0a4cbfa5cad46af +3,64,2,5,330370,319714,381249,211312,167772160,167772160,a0a4cbfa5cad46af +3,128,1,5,687005,573219,588976,399684,671088640,0,6c7a2ea2bc98454c +3,128,2,5,615369,582380,646598,407928,335544320,335544320,6c7a2ea2bc98454c +3,256,1,5,1424505,1033384,1168923,792824,1342177280,0,fe51c755b1336171 +3,256,2,5,1309757,1098398,1209954,801232,671088640,671088640,fe51c755b1336171 diff --git a/kv-service/benchmarks/results/softroce-vm-test-manifest.json b/kv-service/benchmarks/results/softroce-vm-test-manifest.json new file mode 100644 index 0000000..2fcd9de --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-test-manifest.json @@ -0,0 +1,117 @@ +{ + "environment": "soft-roce KVM, not physical HCA", + "receipts": [ + { + "side": "client", + "file": "dual-success-e2e.log", + "status": "passed", + "purpose": "single and dual rail byte equality", + "raw_sha256": "caa6f2a69d8b3c8733a5adc17d03aeb8dfc147d6d4254c22e35942174840ddd0", + "raw_bytes": 166 + }, + { + "side": "client", + "file": "dual-dead-listener-e2e.log", + "status": "passed", + "purpose": "second listener unavailable leaves caller buffer unchanged", + "raw_sha256": "52d920ca094f13d4087fb065ccbc77f914ac41cf132c6ccac7376e8f9af4dcad", + "raw_bytes": 175 + }, + { + "side": "client", + "file": "dual-stale-generation-e2e.log", + "status": "passed", + "purpose": "old descriptor rejected after rewrite", + "raw_sha256": "a91d4512559943a08bb9679bfe71a21d86979675a5c6cb6be3f680f411b165c3", + "raw_bytes": 177 + }, + { + "side": "client", + "file": "dual-cancel-inflight.log", + "status": "passed", + "purpose": "both rails active, cancellation and buffer reuse", + "raw_sha256": "12cd4c32457bb5d034e8ddee0601180a617ba715e62a7289654e7455fcf760db", + "raw_bytes": 187 + }, + { + "side": "client", + "file": "dual-stripe1-corruption.log", + "status": "passed", + "purpose": "stripe one checksum failure on second rail", + "raw_sha256": "6bf6abbd2df12bded04cef0ee377099a54ff8e6d1fb0d5d75702340c94d2dce5", + "raw_bytes": 184 + }, + { + "side": "client", + "file": "dual-late-after-retire-fix.log", + "status": "passed", + "purpose": "first rail complete and second rail late WRITE safe failure", + "raw_sha256": "e4af8b5c70b07b5855a33bb1a66e35fe789c866bb88851c926a0cf2b896ed254", + "raw_bytes": 191 + }, + { + "side": "client", + "file": "dual-fallback-verbs.log", + "status": "passed", + "purpose": "two rails restore one object with slab disabled", + "raw_sha256": "0f8b66e434e26bc8dff2bc6d24d10d1910e9372cf73ee8cd147b6b5bb83a87ad", + "raw_bytes": 1300 + }, + { + "side": "client", + "file": "cq-retire-red.log", + "status": "expected failure", + "purpose": "old client and server reused stale control response", + "raw_sha256": "6c5637e5924edfbc41a8034cd1d3dbe7af2567c45c57271175c02c6145b9f6f9", + "raw_bytes": 531 + }, + { + "side": "client", + "file": "cq-retire-green-client-fix.log", + "status": "passed", + "purpose": "connection retired after control timeout", + "raw_sha256": "c363fc9c87caff7c7b9bf54b55411f78da8b6c9ce3c463073f4cf1cea0c8604c", + "raw_bytes": 176 + }, + { + "side": "client", + "file": "memlock-preflight.log", + "status": "expected resource rejection", + "purpose": "finite RLIMIT_MEMLOCK rejects fourth concurrent registration before dispatch", + "raw_sha256": "05d2d7dd31a05c59e67ec3c498bb993bc775aa459938fd90bb9bd9a73cda9a6e", + "raw_bytes": 594 + }, + { + "side": "client", + "file": "memlock-unlimited-success.log", + "status": "passed", + "purpose": "same 128 MiB by four cell after isolated VM memlock increase", + "raw_sha256": "28686e6231ddced491fbaf1fe9cee89af53543d4d8e12ef98f7aaa7686760286", + "raw_bytes": 935 + }, + { + "side": "server", + "file": "server-dual-late-cq2s.log", + "status": "diagnostic", + "purpose": "second rail CQ deadline 2 seconds and source retained", + "raw_sha256": "9f8605a171f061007699fac3e6c13758292edef17d919dfd8fbac6f31efe5f67", + "raw_bytes": 3492 + }, + { + "side": "server", + "file": "server-cq-retire-green.log", + "status": "diagnostic", + "purpose": "uncertain completion tears down server QP/CQ connection", + "raw_sha256": "23e368667d5562be2cd30a0fabc22f54b2860a0d969f45cdd84d3c049638b59d", + "raw_bytes": 6693 + }, + { + "side": "server", + "file": "server-dual-fallback.log", + "status": "diagnostic", + "purpose": "both listeners entered registered-buffer fallback", + "raw_sha256": "864aabe8c1d5792a7606bab0963c636036ef9c4c89e852554455dabfdaea55dc", + "raw_bytes": 5804 + } + ] +} diff --git a/kv-service/benchmarks/results/softroce-vm-test-receipts.txt b/kv-service/benchmarks/results/softroce-vm-test-receipts.txt new file mode 100644 index 0000000..e4ac442 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-test-receipts.txt @@ -0,0 +1,152 @@ +ContextStore Soft-RoCE two-rail test receipts + +Environment: two isolated Ubuntu 22.04 KVM guests; two independent virtio/tap/bridge RXE paths. +Expected failure is a deliberate red or resource-boundary result; it is not a passed E2E test. +Full raw logs are retained in the local outputs/remote_evidence/softroce_* folders. + +=== client/dual-success-e2e.log | passed | single and dual rail byte equality === + +running 1 test +test same_object_matches_single_and_dual_rail ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 11 filtered out; finished in 1.66s + + +=== client/dual-dead-listener-e2e.log | passed | second listener unavailable leaves caller buffer unchanged === + +running 1 test +test failed_second_rail_does_not_publish_partial_bytes ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 11 filtered out; finished in 0.66s + + +=== client/dual-stale-generation-e2e.log | passed | old descriptor rejected after rewrite === + +running 1 test +test old_generation_is_rejected_without_publishing_bytes ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 11 filtered out; finished in 1.43s + + +=== client/dual-cancel-inflight.log | passed | both rails active, cancellation and buffer reuse === + +running 1 test +test cancellation_during_two_rail_transfer_preserves_reused_buffer ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 10 filtered out; finished in 3.27s + + +=== client/dual-stripe1-corruption.log | passed | stripe one checksum failure on second rail === + +running 1 test +test corrupt_stripe_on_second_rail_cannot_publish_partial_object ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 9 filtered out; finished in 0.33s + + +=== client/dual-late-after-retire-fix.log | passed | first rail complete and second rail late WRITE safe failure === + +running 1 test +test late_second_rail_after_first_completion_cannot_publish_or_corrupt ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 11 filtered out; finished in 9.72s + + +=== client/dual-fallback-verbs.log | passed | two rails restore one object with slab disabled === +route,id=r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153 +route,id=r1,device=rxe_c1,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.32.0.2:55154 +sample,environment=soft-roce,rails=2,iteration=1,bytes=67108864,latency_us=347917,gib_per_s=0.180,xxh3=a0a4cbfa5cad46af +sample,environment=soft-roce,rails=2,iteration=2,bytes=67108864,latency_us=355554,gib_per_s=0.176,xxh3=a0a4cbfa5cad46af +sample,environment=soft-roce,rails=2,iteration=3,bytes=67108864,latency_us=341445,gib_per_s=0.183,xxh3=a0a4cbfa5cad46af +summary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=3,median_us=347917,cpu_user_us=176731,cpu_system_us=246551,peak_rss_kb=211160,rail_bytes=[100663296, 100663296] +rail_stats,id=r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=4,reads_err=0,bytes=134217728,avg_us=274448,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736 +rail_stats,id=r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=4,reads_err=0,bytes=134217728,avg_us=269701,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736 + +=== client/cq-retire-red.log | expected failure | old client and server reused stale control response === + +running 1 test + +thread 'uncertain_completion_retires_old_server_connection' panicked at kv-service/client-rs/tests/rail_read_e2e.rs:446:5: +server must close the old QP/CQ rather than return a stale response +note: run with `RUST_BACKTRACE=1` environment variable to display a backtrace +test uncertain_completion_retires_old_server_connection ... FAILED + +failures: + +failures: + uncertain_completion_retires_old_server_connection + +test result: FAILED. 0 passed; 1 failed; 0 ignored; 0 measured; 11 filtered out; finished in 8.01s + + +=== client/cq-retire-green-client-fix.log | passed | connection retired after control timeout === + +running 1 test +test uncertain_completion_retires_old_server_connection ... ok + +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 11 filtered out; finished in 7.51s + + +=== client/memlock-preflight.log | expected resource rejection | finite RLIMIT_MEMLOCK rejects fourth concurrent registration before dispatch === +route,id=r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153 +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=0,bytes=134217728,latency_us=1505892,xxh3=6c7a2ea2bc98454c +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=1,bytes=134217728,latency_us=1453796,xxh3=6c7a2ea2bc98454c +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=3,bytes=134217728,latency_us=1357929,xxh3=6c7a2ea2bc98454c +Error: rail resource limit: registered bytes exceed configured or process memlock budget (409609831 bytes) +exit_code=1 + +=== client/memlock-unlimited-success.log | passed | same 128 MiB by four cell after isolated VM memlock increase === +route,id=r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153 +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=0,bytes=134217728,latency_us=1829333,xxh3=6c7a2ea2bc98454c +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=1,bytes=134217728,latency_us=1915528,xxh3=6c7a2ea2bc98454c +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=2,bytes=134217728,latency_us=1655936,xxh3=6c7a2ea2bc98454c +sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=3,bytes=134217728,latency_us=1873385,xxh3=6c7a2ea2bc98454c +batch,environment=soft-roce,rails=1,concurrency=4,iteration=1,wall_us=1976160 +summary_concurrent,environment=soft-roce,rails=1,concurrency=4,bytes_per_read=134217728,batches=1,median_request_us=1873385,median_batch_us=1976160,aggregate_gib_per_s=0.253,cpu_user_us=496302,cpu_system_us=570090,peak_rss_kb=1290684,rail_bytes=[536870912] + +=== server/server-dual-late-cq2s.log | diagnostic | second rail CQ deadline 2 seconds and source retained === +2026-10-02T05:06:13.426978Z INFO RDMA client connected: 10.31.0.1:39878 (nic_idx=0) +2026-10-02T05:06:13.427313Z INFO RDMA client connected: 10.32.0.1:33862 (nic_idx=1) +2026-10-02T05:06:13.454721Z WARN delaying stripe-subset WRITEs for fault injection event="rdma_test_pre_write_delay" nic_idx=1 delay_ms=3000 +2026-10-02T05:06:13.579388Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790917572956012522 bytes=33554432 stripes=8 requests=32 alloc_us=4 stream_setup_us=1626 first_io_us=9309 last_io_us=20327 poll_us=106765 total_us=127074 +2026-10-02T05:06:18.456047Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790917572956012522 bytes=33554432 stripes=8 requests=32 alloc_us=3 stream_setup_us=154 first_io_us=3000455 last_io_us=3001451 poll_us=2000091 total_us=5001494 +2026-10-02T05:06:18.456446Z WARN RDMA stripe-subset stream failed without deleting object metadata key=8:rail-e2erail-e2e-1790917572956012522 error=poll final RDMA completions: poll_n timeout after 2000ms, got 0/32 + +=== server/server-cq-retire-green.log | diagnostic | uncertain completion tears down server QP/CQ connection === +2026-10-02T06:00:37.226226Z INFO RDMA client connected: 10.32.0.1:59304 (nic_idx=1) +2026-10-02T06:00:37.252949Z WARN delaying stripe-subset WRITEs for fault injection event="rdma_test_pre_write_delay" nic_idx=1 delay_ms=3000 +2026-10-02T06:00:40.276172Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790920836691090604 bytes=4194304 stripes=1 requests=4 alloc_us=4 stream_setup_us=2023 first_io_us=3002276 last_io_us=3003113 poll_us=22144 total_us=3025263 +2026-10-02T06:00:44.259687Z WARN delaying stripe-subset WRITEs for fault injection event="rdma_test_pre_write_delay" nic_idx=1 delay_ms=3000 +2026-10-02T06:00:49.260707Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790920836691090604 bytes=4194304 stripes=1 requests=4 alloc_us=4 stream_setup_us=1767 first_io_us=3001971 last_io_us=3002769 poll_us=2000013 total_us=5002802 +2026-10-02T06:00:49.261107Z WARN RDMA stripe-subset stream failed without deleting object metadata key=8:rail-e2erail-e2e-1790920836691090604 error=poll final RDMA completions: poll_n timeout after 2000ms, got 0/4 +2026-10-02T06:00:49.261257Z WARN RDMA client 10.32.0.1:59304 (nic_idx=1) disconnected: RDMA connection retired after uncertain stripe-subset completion: poll final RDMA completions: poll_n timeout after 2000ms, got 0/4 +2026-10-02T06:04:52.924240Z INFO RDMA client connected: 10.32.0.1:46986 (nic_idx=1) +2026-10-02T06:04:52.965148Z WARN delaying stripe-subset WRITEs for fault injection event="rdma_test_pre_write_delay" nic_idx=1 delay_ms=3000 +2026-10-02T06:04:57.966983Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790921092474562138 bytes=4194304 stripes=1 requests=4 alloc_us=38 stream_setup_us=2025 first_io_us=3002426 last_io_us=3003833 poll_us=2000036 total_us=5003920 +2026-10-02T06:04:57.967021Z WARN RDMA stripe-subset stream failed without deleting object metadata key=8:rail-e2erail-e2e-1790921092474562138 error=poll final RDMA completions: poll_n timeout after 2000ms, got 0/4 +2026-10-02T06:04:57.967205Z WARN RDMA client 10.32.0.1:46986 (nic_idx=1) disconnected: RDMA connection retired after uncertain stripe-subset completion: poll final RDMA completions: poll_n timeout after 2000ms, got 0/4 +2026-10-02T06:05:36.317812Z INFO RDMA client connected: 10.32.0.1:35712 (nic_idx=1) +2026-10-02T06:05:36.317829Z INFO RDMA client connected: 10.31.0.1:51894 (nic_idx=0) +2026-10-02T06:05:36.373606Z WARN delaying stripe-subset WRITEs for fault injection event="rdma_test_pre_write_delay" nic_idx=1 delay_ms=3000 +2026-10-02T06:05:36.496798Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790921135831858189 bytes=33554432 stripes=8 requests=32 alloc_us=13 stream_setup_us=8272 first_io_us=11138 last_io_us=45542 poll_us=91384 total_us=136914 +2026-10-02T06:05:41.375004Z INFO RDMA_SUBSET_DETAIL key=8:rail-e2erail-e2e-1790921135831858189 bytes=33554432 stripes=8 requests=32 alloc_us=3 stream_setup_us=5996 first_io_us=3006276 last_io_us=3007338 poll_us=2000048 total_us=5007399 +2026-10-02T06:05:41.375048Z WARN RDMA stripe-subset stream failed without deleting object metadata key=8:rail-e2erail-e2e-1790921135831858189 error=poll final RDMA completions: poll_n timeout after 2000ms, got 0/32 +2026-10-02T06:05:41.375230Z WARN RDMA client 10.32.0.1:35712 (nic_idx=1) disconnected: RDMA connection retired after uncertain stripe-subset completion: poll final RDMA completions: poll_n timeout after 2000ms, got 0/32 + +=== server/server-dual-fallback.log | diagnostic | both listeners entered registered-buffer fallback === +2026-10-02T06:09:35.661301Z INFO RDMA client connected: 10.31.0.1:52500 (nic_idx=0) +2026-10-02T06:09:35.661666Z INFO RDMA client connected: 10.32.0.1:55962 (nic_idx=1) +2026-10-02T06:09:35.687778Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:35.687856Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:36.079058Z INFO RDMA client connected: 10.32.0.1:38304 (nic_idx=1) +2026-10-02T06:09:36.079285Z INFO RDMA client connected: 10.31.0.1:44688 (nic_idx=0) +2026-10-02T06:09:36.107049Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:36.107217Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:36.442372Z INFO RDMA client connected: 10.32.0.1:38306 (nic_idx=1) +2026-10-02T06:09:36.442533Z INFO RDMA client connected: 10.31.0.1:44694 (nic_idx=0) +2026-10-02T06:09:36.468840Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:36.471706Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:36.816234Z INFO RDMA client connected: 10.32.0.1:38322 (nic_idx=1) +2026-10-02T06:09:36.816252Z INFO RDMA client connected: 10.31.0.1:44704 (nic_idx=0) +2026-10-02T06:09:36.842161Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable +2026-10-02T06:09:36.843856Z WARN RDMA stripe-subset slab path unavailable; using registered-buffer fallback key=10:rust-benchrxetest0/__combined__ error=RDMA slab is unavailable diff --git a/kv-service/benchmarks/results/softroce-vm-topology.json b/kv-service/benchmarks/results/softroce-vm-topology.json new file mode 100644 index 0000000..2101251 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-topology.json @@ -0,0 +1,99 @@ +{ + "label": "Soft-RoCE RXE through two isolated KVM guest-to-guest virtio Ethernet paths; not physical HCA", + "host": "skv-node1, Intel Xeon Gold 5218R, QEMU 4.2.1", + "guests": [ + { + "role": "client", + "hostname": "cs-railvm-client", + "kernel": "Ubuntu 22.04.5 Linux 5.15.0-194-generic", + "resources": { + "vcpus": 4, + "memory_gib": 4 + }, + "rails": [ + { + "id": "rxe_c0", + "netdev": "enp0s3", + "ipv4": "10.31.0.1/24", + "gid_index": 1, + "gid": "::ffff:10.31.0.1", + "bridge": "csrb0" + }, + { + "id": "rxe_c1", + "netdev": "enp0s4", + "ipv4": "10.32.0.1/24", + "gid_index": 1, + "gid": "::ffff:10.32.0.1", + "bridge": "csrb1" + } + ] + }, + { + "role": "server", + "hostname": "cs-railvm", + "kernel": "Ubuntu 22.04.5 Linux 5.15.0-194-generic", + "resources": { + "vcpus": 4, + "memory_gib": 4 + }, + "rails": [ + { + "id": "rxe_s0", + "netdev": "enp0s3", + "ipv4": "10.31.0.2/24", + "gid_index": 1, + "gid": "::ffff:10.31.0.2", + "bridge": "csrb0", + "listener": "10.31.0.2:55153" + }, + { + "id": "rxe_s1", + "netdev": "enp0s4", + "ipv4": "10.32.0.2/24", + "gid_index": 1, + "gid": "::ffff:10.32.0.2", + "bridge": "csrb1", + "listener": "10.32.0.2:55154" + } + ] + } + ], + "common": { + "rdma_core": "39.0-1", + "source_base": "DaoCloud main b5c6451", + "server_binary_sha256": "f4f2df78e278da5ec6f47a3a0f23f31b0865eaa0709f1b03ca9fc6883c744e01", + "paired_client_source_commit": "52b933d", + "concurrent_client_source_commit": "cd529b5", + "object_layout": "4 MiB stripes across two directories on one server VM virtual disk; verify_stripe_checksums=true", + "storage_boundary": "QEMU qcow2 virtual disk on skv-node1 /home SATA SSD; no independent NVMe, HCA, PCIe or NUMA path" + }, + "counter_probe_64m_5_reads": { + "object_payload_bytes": 335544320, + "rail0": { + "payload_bytes": 167772160, + "server_netdev_tx_before": 13972899714, + "server_netdev_tx_after": 14150221207, + "server_netdev_tx_delta": 177321493, + "client_netdev_rx_before": 13972898798, + "client_netdev_rx_after": 14150220291, + "client_netdev_rx_delta": 177321493 + }, + "rail1": { + "payload_bytes": 167772160, + "server_netdev_tx_before": 4819518568, + "server_netdev_tx_after": 4996867049, + "server_netdev_tx_delta": 177348481, + "client_netdev_rx_before": 4819517492, + "client_netdev_rx_after": 4996865973, + "client_netdev_rx_delta": 177348481 + }, + "caveat": "The netdev counters include protocol overhead and any concurrent traffic; measured around a five-read benchmark with no other test activity." + }, + "limitations": [ + "RXE is software processing; QEMU/virtio/bridges run on one physical host.", + "Per-guest NUMA and PCI topology are virtual and do not validate HCA offload.", + "One virtual disk backs both stripe directories; no multi-disk bandwidth conclusion.", + "Measurements are paired within each size/concurrency cell; trial count is descriptive, not a confidence interval." + ] +} diff --git a/kv-service/benchmarks/softroce_corrupt_restore.py b/kv-service/benchmarks/softroce_corrupt_restore.py new file mode 100644 index 0000000..f777857 --- /dev/null +++ b/kv-service/benchmarks/softroce_corrupt_restore.py @@ -0,0 +1,96 @@ +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import shutil +import subprocess +from pathlib import Path + + +ROOT = Path(os.environ.get("CS_GUEST_ROOT", "/home/railtest")) +BACKUP = ROOT / "evidence/softroce-stripe1.backup" +MANIFEST = ROOT / "evidence/softroce-stripe1-corruption.json" + + +def sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + while chunk := handle.read(1024 * 1024): + digest.update(chunk) + return digest.hexdigest() + + +def stripe_path(namespace: str, key: str) -> Path: + output = subprocess.check_output( + [ + str(ROOT / "bin/cs-meta"), + "--config", + str(ROOT / "evidence/server.toml"), + "--json", + "get", + "--namespace", + namespace, + "--object-key", + key, + ], + text=True, + ) + record = json.loads(output) + path = Path(record["metadata"]["striping"]["chunk_paths"][1]).resolve() + if not path.is_relative_to(ROOT / "data") or not path.name.endswith(".chunk1.bin"): + raise ValueError(f"unexpected test stripe path: {path}") + return path + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("phase", choices=["inject", "restore"]) + parser.add_argument("--namespace", default="rust-bench") + parser.add_argument("--object-key", default="rxetest0/__combined__") + args = parser.parse_args() + target = stripe_path(args.namespace, args.object_key) + if args.phase == "inject": + if BACKUP.exists(): + raise RuntimeError("backup exists; restore before another injection") + shutil.copy2(target, BACKUP) + original = sha256(target) + with target.open("r+b") as handle: + first = handle.read(1) + if not first: + raise RuntimeError("target stripe is empty") + handle.seek(0) + handle.write(bytes([first[0] ^ 0xFF])) + handle.flush() + os.fsync(handle.fileno()) + corrupted = sha256(target) + MANIFEST.write_text( + json.dumps( + { + "namespace": args.namespace, + "object_key": args.object_key, + "stripe_index": 1, + "path": str(target), + "original_sha256": original, + "corrupted_sha256": corrupted, + }, + indent=2, + ) + + "\n" + ) + print(f"injected stripe=1 original={original} corrupted={corrupted}") + else: + if not BACKUP.exists(): + raise RuntimeError("backup is missing") + shutil.copy2(BACKUP, target) + expected = json.loads(MANIFEST.read_text())["original_sha256"] + actual = sha256(target) + if actual != expected: + raise RuntimeError(f"restore mismatch: {actual} != {expected}") + BACKUP.unlink() + print(f"restored stripe=1 sha256={actual}") + + +if __name__ == "__main__": + main() diff --git a/kv-service/configs/server-softroce-vm.toml b/kv-service/configs/server-softroce-vm.toml new file mode 100644 index 0000000..6aaafd4 --- /dev/null +++ b/kv-service/configs/server-softroce-vm.toml @@ -0,0 +1,47 @@ +[api] +listen = "10.31.0.2:55151" +max_connections = 100 + +[cluster] +node_id = "cs-soft-roce-vm-server" +grpc_advertise = "10.31.0.2:55151" +rdma_advertise = "10.31.0.2:55153" +data_nodes = [] + +[storage] +devices = ["/home/railtest/data/rail0", "/home/railtest/data/rail1"] +data_subdir = "contextstore" +striping_threshold = 4194304 +striping_chunk_size = 4194304 +rdma_stream_chunk_size = 1048576 +verify_stripe_checksums = true + +[memory_tier] +capacity_mb = 128 +slab_size_mb = 4 +use_pinned_memory = false + +[io_executor] +kind = "tier_a" +thread_pool_size = 8 +io_uring_depth = 32 + +[router] +strategy = "object_hash" + +[metadata] +redis_url = "redis://127.0.0.1:6388/" +redis_key_prefix = "contextstore:soft-roce-vm:" +redis_connect_timeout_ms = 1000 +redis_command_timeout_ms = 1000 + +[gc] +enabled = false +interval_seconds = 300 +grace_seconds = 600 +max_tasks_per_run = 1000 +task_lease_seconds = 300 + +[metrics] +enabled = false +listen = "127.0.0.1:55190" diff --git a/kv-service/deploy/softroce-vm/build-verbs.sh b/kv-service/deploy/softroce-vm/build-verbs.sh new file mode 100755 index 0000000..7fc26de --- /dev/null +++ b/kv-service/deploy/softroce-vm/build-verbs.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash +set -euo pipefail + +repo_root=$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd) +cd "$repo_root" +cargo build --locked --release -p contextstore-server --features rdma \ + --bin contextstore-server --bin cs-meta +cargo build --locked --release -p contextstore-client-rs --features rdma \ + --bin cs-bench --bin cs-rail-read-bench +cargo test --locked --release -p contextstore-client-rs --features rdma \ + --test rail_read_e2e --no-run diff --git a/kv-service/deploy/softroce-vm/deploy-verbs.sh b/kv-service/deploy/softroce-vm/deploy-verbs.sh new file mode 100755 index 0000000..1b7d155 --- /dev/null +++ b/kv-service/deploy/softroce-vm/deploy-verbs.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Copy Linux x86_64 binaries to two isolated test guests via an SSH config. +repo_root=$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd) +bin_dir=${CS_BIN_DIR:-$repo_root/target/release} +ssh_config=${CS_VM_SSH_CONFIG:?set CS_VM_SSH_CONFIG to SSH aliases for both guests} +server_alias=${CS_VM_SERVER_ALIAS:-cs-railvm-server} +client_alias=${CS_VM_CLIENT_ALIAS:-cs-railvm-client} +guest_root=${CS_GUEST_ROOT:-/home/railtest} + +for guest in "$server_alias" "$client_alias"; do + ssh -F "$ssh_config" "$guest" "mkdir -p '$guest_root/bin' '$guest_root/evidence' '$guest_root/data/rail0' '$guest_root/data/rail1'" +done +scp -F "$ssh_config" "$bin_dir/contextstore-server" "$bin_dir/cs-meta" \ + "$server_alias:$guest_root/bin/" +scp -F "$ssh_config" "$bin_dir/cs-bench" "$bin_dir/cs-rail-read-bench" \ + "$client_alias:$guest_root/bin/" +scp -F "$ssh_config" \ + "$repo_root/kv-service/configs/server-softroce-vm.toml" \ + "$server_alias:$guest_root/evidence/server.toml" +scp -F "$ssh_config" \ + "$repo_root/kv-service/deploy/softroce-vm/start-guest-server.sh" \ + "$repo_root/kv-service/deploy/softroce-vm/stop-guest-server.sh" \ + "$server_alias:$guest_root/" +scp -F "$ssh_config" \ + "$repo_root/kv-service/benchmarks/collect_rail_verbs.py" \ + "$repo_root/kv-service/benchmarks/collect_rail_concurrent.py" \ + "$client_alias:$guest_root/" +if [[ -n "${CS_E2E_BIN:-}" ]]; then + scp -F "$ssh_config" "$CS_E2E_BIN" \ + "$client_alias:$guest_root/bin/rail_read_e2e" +fi diff --git a/kv-service/deploy/softroce-vm/install-guest-prereqs.sh b/kv-service/deploy/softroce-vm/install-guest-prereqs.sh new file mode 100755 index 0000000..fe29968 --- /dev/null +++ b/kv-service/deploy/softroce-vm/install-guest-prereqs.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Run only inside a fresh Ubuntu 22.04 test guest, before setup-guest-paths.sh. +sudo env DEBIAN_FRONTEND=noninteractive apt-get update -qq +sudo env DEBIAN_FRONTEND=noninteractive apt-get install -y -qq \ + --no-install-recommends \ + "linux-modules-extra-$(uname -r)" rdma-core libibverbs1 \ + ibverbs-providers ibverbs-utils redis-server +modinfo -n rdma_rxe diff --git a/kv-service/deploy/softroce-vm/prepare-pair.sh b/kv-service/deploy/softroce-vm/prepare-pair.sh new file mode 100755 index 0000000..fdcf337 --- /dev/null +++ b/kv-service/deploy/softroce-vm/prepare-pair.sh @@ -0,0 +1,60 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Prepare two separate Ubuntu cloud-image overlays and NoCloud seeds. +# This reads one public SSH key; private keys never enter the repository. +vm_dir=${CS_VM_DIR:?set CS_VM_DIR to an empty test-only directory} +public_key=${CS_VM_SSH_PUBLIC_KEY:?set CS_VM_SSH_PUBLIC_KEY to a public key path} +guest_user=${CS_VM_GUEST_USER:-railtest} +if [[ ! -r "$public_key" ]]; then + echo "Public SSH key is not readable: $public_key" >&2 + exit 1 +fi +mkdir -p "$vm_dir" +cd "$vm_dir" + +image=jammy-server-cloudimg-amd64.img +base_url=https://cloud-images.ubuntu.com/jammy/current +if [[ ! -f "$image" ]]; then + curl -fsSL --retry 3 "$base_url/$image" -o "$image" +fi +curl -fsSL --retry 3 "$base_url/SHA256SUMS" -o SHA256SUMS +sha256sum -c SHA256SUMS --ignore-missing | grep -qx "$image: OK" + +for role in server client; do + disk="disk-${role}.qcow2" + if [[ ! -f "$disk" ]]; then + qemu-img create -q -f qcow2 -F qcow2 -b "$image" "$disk" 12G + fi + seed_dir="seed-${role}" + mkdir -p "$seed_dir" + cat > "$seed_dir/user-data" < "$seed_dir/meta-data" <&2 + exit 2 + ;; +esac + +sudo ip addr add "$ip0" dev enp0s3 +sudo ip addr add "$ip1" dev enp0s4 +sudo ip link set enp0s3 up +sudo ip link set enp0s4 up +sudo modprobe rdma_rxe +sudo rdma link add "$device0" type rxe netdev enp0s3 +sudo rdma link add "$device1" type rxe netdev enp0s4 +rdma link show +ibv_devices diff --git a/kv-service/deploy/softroce-vm/setup-host-bridges.sh b/kv-service/deploy/softroce-vm/setup-host-bridges.sh new file mode 100755 index 0000000..77aecf0 --- /dev/null +++ b/kv-service/deploy/softroce-vm/setup-host-bridges.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash +set -euo pipefail + +tap_user=${CS_VM_TAP_USER:-$(id -un)} + +for name in csrb0 csrb1 csrs0 csrc0 csrs1 csrc1; do + if ip link show dev "$name" >/dev/null 2>&1; then + echo "Refusing to replace existing interface: $name" >&2 + exit 1 + fi +done + +sudo ip link add name csrb0 type bridge +sudo ip link add name csrb1 type bridge +for name in csrs0 csrc0 csrs1 csrc1; do + sudo ip tuntap add dev "$name" mode tap user "$tap_user" +done +sudo ip link set csrs0 master csrb0 +sudo ip link set csrc0 master csrb0 +sudo ip link set csrs1 master csrb1 +sudo ip link set csrc1 master csrb1 +for name in csrb0 csrb1 csrs0 csrc0 csrs1 csrc1; do + sudo ip link set "$name" up +done +ip -br link show csrb0 +ip -br link show csrb1 +bridge link show | grep -E 'csrs[01]|csrc[01]' diff --git a/kv-service/deploy/softroce-vm/ssh_config.example b/kv-service/deploy/softroce-vm/ssh_config.example new file mode 100644 index 0000000..f20bc2f --- /dev/null +++ b/kv-service/deploy/softroce-vm/ssh_config.example @@ -0,0 +1,20 @@ +# Copy outside the repository, then replace KVM_HOST_ALIAS and PRIVATE_KEY_PATH. +Host cs-railvm-server + HostName 127.0.0.1 + User railtest + Port 22222 + ProxyCommand ssh KVM_HOST_ALIAS -W %h:%p + IdentityFile PRIVATE_KEY_PATH + IdentitiesOnly yes + BatchMode yes + StrictHostKeyChecking accept-new + +Host cs-railvm-client + HostName 127.0.0.1 + User railtest + Port 22223 + ProxyCommand ssh KVM_HOST_ALIAS -W %h:%p + IdentityFile PRIVATE_KEY_PATH + IdentitiesOnly yes + BatchMode yes + StrictHostKeyChecking accept-new diff --git a/kv-service/deploy/softroce-vm/start-guest-server.sh b/kv-service/deploy/softroce-vm/start-guest-server.sh new file mode 100755 index 0000000..7621b90 --- /dev/null +++ b/kv-service/deploy/softroce-vm/start-guest-server.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +set -euo pipefail + +root=${CS_GUEST_ROOT:-/home/railtest} +redis_session=cs-softroce-redis +server_session=cs-softroce-server +slab_mb=${CS_RDMA_SLAB_MB:-128} +write_delay_ms=${CS_RDMA_TEST_PRE_WRITE_DELAY_MS:-0} +write_delay_nic=${CS_RDMA_TEST_PRE_WRITE_NIC_IDX:-0} +cq_timeout_ms=${CS_RDMA_CQ_TIMEOUT_MS:-30000} +mkdir -p "$root/data/redis" "$root/data/rail0" "$root/data/rail1" "$root/evidence" + +if ! tmux has-session -t "$redis_session" 2>/dev/null; then + tmux new-session -d -s "$redis_session" \ + "exec redis-server --bind 127.0.0.1 --port 6388 --save '' --appendonly no --dir '$root/data/redis' >> '$root/evidence/redis.log' 2>&1" +fi +for _ in $(seq 1 30); do + if redis-cli -h 127.0.0.1 -p 6388 PING 2>/dev/null | grep -qx PONG; then + break + fi + sleep 0.2 +done +redis-cli -h 127.0.0.1 -p 6388 PING + +if tmux has-session -t "$server_session" 2>/dev/null; then + echo "Server session already exists: $server_session" >&2 + exit 1 +fi +tmux new-session -d -s "$server_session" \ + "CS_RDMA_DEVICES='rxe_s0:10.31.0.2:55153:1,rxe_s1:10.32.0.2:55154:1' CS_RDMA_SLAB_MB='$slab_mb' CS_RDMA_CQ_TIMEOUT_MS='$cq_timeout_ms' CS_RDMA_TEST_PRE_WRITE_DELAY_MS='$write_delay_ms' CS_RDMA_TEST_PRE_WRITE_NIC_IDX='$write_delay_nic' exec '$root/bin/contextstore-server' --config '$root/evidence/server.toml' >> '$root/evidence/server.log' 2>&1" +sleep 2 +tmux has-session -t "$server_session" +ss -ltn | grep -E ':(55151|55153|55154)\b' diff --git a/kv-service/deploy/softroce-vm/start-pair.sh b/kv-service/deploy/softroce-vm/start-pair.sh new file mode 100755 index 0000000..e62cafd --- /dev/null +++ b/kv-service/deploy/softroce-vm/start-pair.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash +set -euo pipefail + +vm_dir=${CS_VM_DIR:?set CS_VM_DIR to the prepared VM directory} +server_session=cs-railvm-server-20261002 +client_session=cs-railvm-client-20261002 +cd "$vm_dir" +if tmux has-session -t "$server_session" 2>/dev/null; then + echo "Server VM already running" >&2 + exit 1 +fi +if tmux has-session -t "$client_session" 2>/dev/null; then + echo "Client VM already running" >&2 + exit 1 +fi + +tmux new-session -d -s "$server_session" \ + "cd '$vm_dir' && exec qemu-system-x86_64 -enable-kvm -cpu host -smp 4 -m 4096 -machine q35 -drive if=pflash,format=raw,readonly=on,file=/usr/share/OVMF/OVMF_CODE.fd -drive if=pflash,format=raw,file=OVMF_VARS-server.fd -drive file=disk-server.qcow2,if=virtio,format=qcow2 -drive file=seed-server.iso,media=cdrom,if=ide -netdev user,id=mgmt,hostfwd=tcp:127.0.0.1:22222-:22 -device virtio-net-pci,netdev=mgmt,mac=52:54:00:12:34:56 -netdev tap,id=rail0,ifname=csrs0,script=no,downscript=no -device virtio-net-pci,netdev=rail0,mac=52:54:00:31:00:02 -netdev tap,id=rail1,ifname=csrs1,script=no,downscript=no -device virtio-net-pci,netdev=rail1,mac=52:54:00:32:00:02 -nographic -serial mon:stdio > qemu-server.log 2>&1" +sleep 2 +tmux has-session -t "$server_session" +ss -ltn | grep ':22222\b' + +tmux new-session -d -s "$client_session" \ + "cd '$vm_dir' && exec qemu-system-x86_64 -enable-kvm -cpu host -smp 4 -m 4096 -machine q35 -drive if=pflash,format=raw,readonly=on,file=/usr/share/OVMF/OVMF_CODE.fd -drive if=pflash,format=raw,file=OVMF_VARS-client.fd -drive file=disk-client.qcow2,if=virtio,format=qcow2 -drive file=seed-client.iso,media=cdrom,if=ide -netdev user,id=mgmt,hostfwd=tcp:127.0.0.1:22223-:22 -device virtio-net-pci,netdev=mgmt,mac=52:54:00:12:34:57 -netdev tap,id=rail0,ifname=csrc0,script=no,downscript=no -device virtio-net-pci,netdev=rail0,mac=52:54:00:31:00:01 -netdev tap,id=rail1,ifname=csrc1,script=no,downscript=no -device virtio-net-pci,netdev=rail1,mac=52:54:00:32:00:01 -nographic -serial mon:stdio > qemu-client.log 2>&1" +sleep 2 +tmux has-session -t "$client_session" +ss -ltn | grep -E ':(22222|22223)\b' diff --git a/kv-service/deploy/softroce-vm/stop-guest-server.sh b/kv-service/deploy/softroce-vm/stop-guest-server.sh new file mode 100755 index 0000000..d0ce4fb --- /dev/null +++ b/kv-service/deploy/softroce-vm/stop-guest-server.sh @@ -0,0 +1,8 @@ +#!/usr/bin/env bash +set -euo pipefail + +for session in cs-softroce-server cs-softroce-redis; do + if tmux has-session -t "$session" 2>/dev/null; then + tmux kill-session -t "$session" + fi +done diff --git a/kv-service/deploy/softroce-vm/stop-pair.sh b/kv-service/deploy/softroce-vm/stop-pair.sh new file mode 100755 index 0000000..b96e4bd --- /dev/null +++ b/kv-service/deploy/softroce-vm/stop-pair.sh @@ -0,0 +1,10 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Prefer guest `sudo poweroff` over this forced stop to preserve qcow2 state. + +for session in cs-railvm-client-20261002 cs-railvm-server-20261002; do + if tmux has-session -t "$session" 2>/dev/null; then + tmux kill-session -t "$session" + fi +done diff --git a/kv-service/deploy/softroce-vm/teardown-guest-paths.sh b/kv-service/deploy/softroce-vm/teardown-guest-paths.sh new file mode 100755 index 0000000..e6cb56a --- /dev/null +++ b/kv-service/deploy/softroce-vm/teardown-guest-paths.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash +set -euo pipefail + +role=${1:?usage: teardown_railvm_paths.sh server|client} +case "$role" in + server) device0=rxe_s0; device1=rxe_s1; ip0=10.31.0.2/24; ip1=10.32.0.2/24 ;; + client) device0=rxe_c0; device1=rxe_c1; ip0=10.31.0.1/24; ip1=10.32.0.1/24 ;; + *) exit 2 ;; +esac +sudo rdma link delete "$device0" +sudo rdma link delete "$device1" +sudo ip addr del "$ip0" dev enp0s3 +sudo ip addr del "$ip1" dev enp0s4 +sudo ip link set enp0s3 down +sudo ip link set enp0s4 down diff --git a/kv-service/deploy/softroce-vm/teardown-host-bridges.sh b/kv-service/deploy/softroce-vm/teardown-host-bridges.sh new file mode 100755 index 0000000..75bd366 --- /dev/null +++ b/kv-service/deploy/softroce-vm/teardown-host-bridges.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +set -euo pipefail + +for name in csrs0 csrc0 csrs1 csrc1; do + if ip link show dev "$name" >/dev/null 2>&1; then + sudo ip link delete "$name" + fi +done +for name in csrb0 csrb1; do + if ip link show dev "$name" >/dev/null 2>&1; then + sudo ip link delete "$name" + fi +done From 81f437df556fab4c8c6fc8c5533f8d78136f624d Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 14:17:35 +0800 Subject: [PATCH 16/20] Clarify stale-response regression provenance The red RXE test exercised an old client against a server with CQ retirement; it showed that client-side control timeout also requires QP retirement. Correct the public test receipt wording without changing the recorded output or hashes. --- kv-service/benchmarks/results/softroce-vm-test-manifest.json | 2 +- kv-service/benchmarks/results/softroce-vm-test-receipts.txt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/kv-service/benchmarks/results/softroce-vm-test-manifest.json b/kv-service/benchmarks/results/softroce-vm-test-manifest.json index 2fcd9de..a175bce 100644 --- a/kv-service/benchmarks/results/softroce-vm-test-manifest.json +++ b/kv-service/benchmarks/results/softroce-vm-test-manifest.json @@ -61,7 +61,7 @@ "side": "client", "file": "cq-retire-red.log", "status": "expected failure", - "purpose": "old client and server reused stale control response", + "purpose": "old client accepted a stale reply after its GET timed out; server CQ retirement alone did not prevent that", "raw_sha256": "6c5637e5924edfbc41a8034cd1d3dbe7af2567c45c57271175c02c6145b9f6f9", "raw_bytes": 531 }, diff --git a/kv-service/benchmarks/results/softroce-vm-test-receipts.txt b/kv-service/benchmarks/results/softroce-vm-test-receipts.txt index e4ac442..4f2c8b1 100644 --- a/kv-service/benchmarks/results/softroce-vm-test-receipts.txt +++ b/kv-service/benchmarks/results/softroce-vm-test-receipts.txt @@ -62,7 +62,7 @@ summary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=3,median_us= rail_stats,id=r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=4,reads_err=0,bytes=134217728,avg_us=274448,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736 rail_stats,id=r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=4,reads_err=0,bytes=134217728,avg_us=269701,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736 -=== client/cq-retire-red.log | expected failure | old client and server reused stale control response === +=== client/cq-retire-red.log | expected failure | old client accepted a stale reply after GET timeout; server CQ retirement alone was insufficient === running 1 test From c439c59a942baf4143089c452f7e7d9cad2a9473 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 14:48:00 +0800 Subject: [PATCH 17/20] Retire legacy RDMA GET connections when WRITE completion is uncertain Pin slab and per-chunk registered sources until QP destruction on post or CQ errors, preventing cache-miss fallback from reusing an uncertain queue. Add a post-handshake GET timeout regression verified on two isolated RXE rails and record the raw receipt hashes. --- README.md | 18 +++- .../results/softroce-vm-test-manifest.json | 16 +++ .../results/softroce-vm-test-receipts.txt | 8 ++ kv-service/client-rs/src/rdma.rs | 8 ++ kv-service/client-rs/tests/rail_read_e2e.rs | 29 +++++ kv-service/server/src/rdma/qp.rs | 43 ++++++-- kv-service/server/src/rdma/server.rs | 102 +++++++++++++++--- 7 files changed, 194 insertions(+), 30 deletions(-) diff --git a/README.md b/README.md index 0a5de75..aadcdfb 100644 --- a/README.md +++ b/README.md @@ -354,11 +354,15 @@ when its registered slab cannot provide staging space. A fallback WRITE with uncertain completion retains its source and MR until its QP is destroyed. Uncertain CQ completion terminates the server connection rather than reusing that CQ for another request; a client GET control error likewise destroys its -QP so a late reply cannot be mistaken for a new request's response. +QP so a late reply cannot be mistaken for a new request's response. This also +applies to the older complete-object GET: cache-hit slab pins, cache-miss slab +extents, and per-chunk fallback MRs remain alive until their QP is destroyed +when a posted WRITE has uncertain completion. A cache-miss CQ error cannot +enter fallback on the same QP/CQ. For hardware-independent scheduling and failure checks, run `cargo test --manifest-path kv-service/client-rs/Cargo.toml --features rdma -rail_read::tests`. The ignored `rail_read_e2e` suite contains five single +rail_read::tests`. The ignored `rail_read_e2e` suite contains six single real-Rail checks and seven dual-Rail/control-connection checks. Select a test with `--ignored --exact --nocapture` and configure `CS_RAIL_COORDINATOR`, `CS_RAIL_LISTENER0/1`, `CS_RAIL_DEVICE0/1`, and `CS_RAIL_GID0/1` as needed. @@ -367,9 +371,13 @@ On an isolated server only, `CS_RDMA_TEST_PRE_WRITE_DELAY_MS=3000` delays stripe-subset WRITEs for the late-completion test; the delay is bounded to five seconds. Add `CS_RDMA_TEST_PRE_WRITE_NIC_IDX=1` to delay only the second listener and exercise partial completion. `CS_RDMA_CQ_TIMEOUT_MS` bounds the -stripe-subset CQ poll between 100 and 30,000 ms (default 30,000); use 2,000 -ms for isolated late-WRITE injection. Unset both fault-injection variables -after testing. A physical-stripe corruption +stripe-subset and legacy GET CQ polls between 100 and 30,000 ms (default +30,000); use 2,000 ms for isolated late-WRITE injection. Unset both +fault-injection variables after testing. With `CS_FORCE_DISK_READ=1` on an +isolated server, the ignored `legacy_get_timeout_retires_qp_before_buffer_reuse` +test completes the QP handshake first, then applies a 1 ms GET deadline. The +matching RXE receipt records a 2 s server CQ timeout, connection retirement, +and an unchanged reused destination after six seconds. A physical-stripe corruption test additionally requires checksum verification enabled before writing its object and a reversible fault injection into one test-only stripe file. The ignored `software_only_mock_benchmark` exercises scheduling and memory diff --git a/kv-service/benchmarks/results/softroce-vm-test-manifest.json b/kv-service/benchmarks/results/softroce-vm-test-manifest.json index a175bce..006a665 100644 --- a/kv-service/benchmarks/results/softroce-vm-test-manifest.json +++ b/kv-service/benchmarks/results/softroce-vm-test-manifest.json @@ -73,6 +73,22 @@ "raw_sha256": "c363fc9c87caff7c7b9bf54b55411f78da8b6c9ce3c463073f4cf1cea0c8604c", "raw_bytes": 176 }, + { + "side": "client", + "file": "legacy-get-timeout-safe.log", + "status": "passed", + "purpose": "legacy full-object GET retires QP on 1ms GET-only client timeout after handshake; reused destination stays unchanged after six seconds", + "raw_sha256": "092cad25d8e73201e5f96b099e5b5144d11a9af9511458671762d2c22ad07b1d", + "raw_bytes": 397 + }, + { + "side": "server", + "file": "server-legacy-timeout-safe.log", + "status": "diagnostic", + "purpose": "legacy slab cache-miss GET retires connection after 2s CQ timeout with 0 of 64 completions; no per-chunk fallback", + "raw_sha256": "33678982a48e3572936e3f644bbfce76a107d8662affa72fd8b3320f386f264a", + "raw_bytes": 2475 + }, { "side": "client", "file": "memlock-preflight.log", diff --git a/kv-service/benchmarks/results/softroce-vm-test-receipts.txt b/kv-service/benchmarks/results/softroce-vm-test-receipts.txt index 4f2c8b1..f930ac0 100644 --- a/kv-service/benchmarks/results/softroce-vm-test-receipts.txt +++ b/kv-service/benchmarks/results/softroce-vm-test-receipts.txt @@ -87,6 +87,14 @@ test uncertain_completion_retires_old_server_connection ... ok test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 11 filtered out; finished in 7.51s +=== client/legacy-get-timeout-safe.log | passed | handshake completes before 1ms GET deadline; reused destination stays unchanged for six seconds === +running 1 test +test legacy_get_timeout_retires_qp_before_buffer_reuse ... ok +test result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 12 filtered out; finished in 6.67s + +=== server/server-legacy-timeout-safe.log | diagnostic | old complete-object GET retires QP after uncertain CQ completion; no same-QP fallback === +2026-10-02T06:45:21.385603Z WARN RDMA client 10.31.0.1:53048 (nic_idx=0) disconnected: RDMA connection retired after uncertain legacy GET: poll RDMA write completion window: poll_n timeout after 2000ms, got 0/64 + === client/memlock-preflight.log | expected resource rejection | finite RLIMIT_MEMLOCK rejects fourth concurrent registration before dispatch === route,id=r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153 sample,environment=soft-roce,rails=1,concurrency=4,batch=1,worker=0,bytes=134217728,latency_us=1505892,xxh3=6c7a2ea2bc98454c diff --git a/kv-service/client-rs/src/rdma.rs b/kv-service/client-rs/src/rdma.rs index e9b5ffa..40f18fb 100644 --- a/kv-service/client-rs/src/rdma.rs +++ b/kv-service/client-rs/src/rdma.rs @@ -372,6 +372,14 @@ impl RdmaClient { } } + /// Change the control-channel deadline after the QP handshake has completed. + /// This also lets callers bound a GET independently of connection setup. + pub fn set_io_timeout(&mut self, timeout: Duration) -> Result<()> { + self.stream.set_read_timeout(Some(timeout))?; + self.stream.set_write_timeout(Some(timeout))?; + Ok(()) + } + fn retire_after_control_error(&mut self) { let _ = self.stream.shutdown(std::net::Shutdown::Both); if let Some(qp) = self.qp.take() { diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs index 74bc402..a261efe 100644 --- a/kv-service/client-rs/tests/rail_read_e2e.rs +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -414,6 +414,35 @@ async fn late_second_rail_after_first_completion_cannot_publish_or_corrupt() { .all(|snapshot| snapshot.inflight_requests == 0 && snapshot.registered_bytes == 0)); } +#[tokio::test] +#[ignore = "requires an isolated RXE server with CS_FORCE_DISK_READ=1 and a 64 MiB object"] +async fn legacy_get_timeout_retires_qp_before_buffer_reuse() { + let (_client, key, payload, advertised) = seeded_object().await; + let listener = setting("CS_RAIL_LISTENER0", &advertised); + let device = setting("CS_RAIL_DEVICE0", "rxe_c0"); + let gid = setting("CS_RAIL_GID0", "1") + .parse::() + .expect("GID index"); + let mut rdma = RdmaClient::connect(RdmaClientConfig::new(listener, device).with_gid_index(gid)) + .expect("connect legacy GET"); + rdma.set_io_timeout(Duration::from_millis(1)) + .expect("set GET deadline after handshake"); + let mut destination = vec![0xA5; payload.len()]; + let registered = rdma + .register_buffer(&mut destination) + .expect("register buffer"); + assert!(rdma.get_into("rail-e2e", &key, ®istered, 0).is_err()); + assert!( + rdma.get_into("rail-e2e", &key, ®istered, 0).is_err(), + "failed legacy GET must not reuse the old QP" + ); + drop(rdma); + drop(registered); + destination.fill(0x33); + tokio::time::sleep(Duration::from_secs(6)).await; + assert!(destination.iter().all(|byte| *byte == 0x33)); +} + #[tokio::test] #[ignore = "requires a three-second delay on server nic_idx=1 and a two-second CQ deadline"] async fn uncertain_completion_retires_old_server_connection() { diff --git a/kv-service/server/src/rdma/qp.rs b/kv-service/server/src/rdma/qp.rs index 905a60c..18ff625 100644 --- a/kv-service/server/src/rdma/qp.rs +++ b/kv-service/server/src/rdma/qp.rs @@ -6,7 +6,7 @@ use anyhow::{anyhow, Result}; use prost::bytes::Bytes; use rdma_sys::*; use std::ptr::{self, NonNull}; -use std::sync::Mutex; +use std::sync::{Arc, Mutex}; /// An RC QP (Reliable Connection Queue Pair). /// @@ -22,6 +22,7 @@ pub struct RcQp { // released only after Drop destroys the QP. retired_writes: Mutex>, retired_extents: Mutex>, + retired_pins: Mutex>>, /// Local QP info, sent to remote over the control plane pub local: QpInfo, } @@ -55,7 +56,9 @@ impl QpInfo { let mut buf = [0u8; 24]; buf[0..4].copy_from_slice(&self.qpn.to_le_bytes()); buf[4..8].copy_from_slice(&self.psn.to_le_bytes()); - unsafe { buf[8..24].copy_from_slice(&self.gid.raw[..]); } + unsafe { + buf[8..24].copy_from_slice(&self.gid.raw[..]); + } buf } @@ -63,7 +66,9 @@ impl QpInfo { let qpn = u32::from_le_bytes(buf[0..4].try_into().unwrap()); let psn = u32::from_le_bytes(buf[4..8].try_into().unwrap()); let mut gid: ibv_gid = unsafe { std::mem::zeroed() }; - unsafe { gid.raw[..].copy_from_slice(&buf[8..24]); } + unsafe { + gid.raw[..].copy_from_slice(&buf[8..24]); + } Self { qpn, psn, gid } } } @@ -93,8 +98,9 @@ impl RcQp { sq_sig_all: 0, // do not signal every wr; caller controls explicitly }; let qp_raw = ibv_create_qp(ctx.pd.as_ptr(), &mut attr); - let qp = NonNull::new(qp_raw) - .ok_or_else(|| anyhow!("ibv_create_qp failed: {}", std::io::Error::last_os_error()))?; + let qp = NonNull::new(qp_raw).ok_or_else(|| { + anyhow!("ibv_create_qp failed: {}", std::io::Error::last_os_error()) + })?; // Pick a random PSN (Packet Serial Number). Cryptographic randomness not required. // Use the low 24 bits of the timestamp. @@ -116,6 +122,7 @@ impl RcQp { qp, retired_writes: Mutex::new(Vec::new()), retired_extents: Mutex::new(Vec::new()), + retired_pins: Mutex::new(Vec::new()), local, }) } @@ -131,7 +138,8 @@ impl RcQp { // Allow remote WRITE/READ on our MR (direction here is server WRITE to client, but symmetric permissions ease debugging) attr.qp_access_flags = (ibv_access_flags::IBV_ACCESS_LOCAL_WRITE.0 | ibv_access_flags::IBV_ACCESS_REMOTE_WRITE.0 - | ibv_access_flags::IBV_ACCESS_REMOTE_READ.0) as i32 as u32; + | ibv_access_flags::IBV_ACCESS_REMOTE_READ.0) + as i32 as u32; let mask = ibv_qp_attr_mask::IBV_QP_STATE | ibv_qp_attr_mask::IBV_QP_PKEY_INDEX @@ -139,7 +147,11 @@ impl RcQp { | ibv_qp_attr_mask::IBV_QP_ACCESS_FLAGS; let rc = ibv_modify_qp(self.qp.as_ptr(), &mut attr, mask.0 as i32); if rc != 0 { - return Err(anyhow!("modify_qp -> INIT failed: rc={} errno={}", rc, std::io::Error::last_os_error())); + return Err(anyhow!( + "modify_qp -> INIT failed: rc={} errno={}", + rc, + std::io::Error::last_os_error() + )); } Ok(()) } @@ -177,7 +189,11 @@ impl RcQp { | ibv_qp_attr_mask::IBV_QP_MIN_RNR_TIMER; let rc = ibv_modify_qp(self.qp.as_ptr(), &mut attr, mask.0 as i32); if rc != 0 { - return Err(anyhow!("modify_qp -> RTR failed: rc={} errno={}", rc, std::io::Error::last_os_error())); + return Err(anyhow!( + "modify_qp -> RTR failed: rc={} errno={}", + rc, + std::io::Error::last_os_error() + )); } Ok(()) } @@ -202,7 +218,11 @@ impl RcQp { | ibv_qp_attr_mask::IBV_QP_MAX_QP_RD_ATOMIC; let rc = ibv_modify_qp(self.qp.as_ptr(), &mut attr, mask.0 as i32); if rc != 0 { - return Err(anyhow!("modify_qp -> RTS failed: rc={} errno={}", rc, std::io::Error::last_os_error())); + return Err(anyhow!( + "modify_qp -> RTS failed: rc={} errno={}", + rc, + std::io::Error::last_os_error() + )); } Ok(()) } @@ -353,6 +373,11 @@ impl RcQp { pub fn retain_uncertain_extent(&self, extent: SlabExtent) { self.retired_extents.lock().unwrap().push(extent); } + + /// Keep a cached slab entry pinned until the QP can no longer read it. + pub fn retain_uncertain_pin(&self, pin: Arc) { + self.retired_pins.lock().unwrap().push(pin); + } } fn wc_status_str(status: u32) -> &'static str { diff --git a/kv-service/server/src/rdma/server.rs b/kv-service/server/src/rdma/server.rs index c7047cc..b74d7a7 100644 --- a/kv-service/server/src/rdma/server.rs +++ b/kv-service/server/src/rdma/server.rs @@ -16,7 +16,7 @@ //! ``` use crate::metadata::{BlockMeta, StripingInfo}; -use crate::rdma::context::RdmaContext; +use crate::rdma::context::{MemRegion, RdmaContext}; use crate::rdma::qp::RcQp; use crate::rdma::slab::{SlabExtent, SlabPlacement}; use crate::rdma::wire::{ @@ -29,6 +29,7 @@ use crate::rdma::wire::{ use crate::router::ObjectKey; use crate::KVServiceContext; use anyhow::{anyhow, Result}; +use prost::bytes::Bytes; use rdma_sys::ibv_access_flags; use std::net::{TcpListener, TcpStream}; use std::ptr::NonNull; @@ -429,6 +430,11 @@ fn handle_client( (false, 0u64, 0u32) } Err(e) => { + if e.downcast_ref::().is_some() { + // A posted WRITE may still read its slab extent. The source is + // pinned by the QP; return now so neither the CQ nor QP is reused. + return Err(e); + } // Slab path failed (slab full / I/O error) → fall back to the old path tracing::warn!( "RDMA GET slab fast path failed, fallback to per-chunk reg_mr: {}", @@ -548,6 +554,21 @@ fn handle_client( /// (no leak) whenever `handle_client` exits (client BYE / protocol error / I/O error). struct CqGuard(NonNull); +#[derive(Debug)] +struct RetireLegacyGet(String); + +impl std::fmt::Display for RetireLegacyGet { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!( + f, + "RDMA connection retired after uncertain legacy GET: {}", + self.0 + ) + } +} + +impl std::error::Error for RetireLegacyGet {} + impl Drop for CqGuard { fn drop(&mut self) { unsafe { @@ -594,7 +615,7 @@ fn serve_get_slab( while offset < total { let len = (total - offset).min(MAX_WRITE_BYTES); let signaled = idx + 1 == n_writes; // Only signal on the last WRITE (RC guarantees prior completions) - qp.post_write( + if let Err(error) = qp.post_write( idx, src_base + offset, lkey, @@ -602,12 +623,18 @@ fn serve_get_slab( dst_rkey, len as u32, signaled, - )?; + ) { + qp.retain_uncertain_pin(placement._pin.clone()); + return Err(anyhow!(RetireLegacyGet(error.to_string()))); + } offset += len; idx += 1; } let t_poll_start = std::time::Instant::now(); - RcQp::poll_n(client_cq, 1)?; + if let Err(error) = RcQp::poll_n_timeout(client_cq, 1, subset_cq_timeout()) { + qp.retain_uncertain_pin(placement._pin.clone()); + return Err(anyhow!(RetireLegacyGet(error.to_string()))); + } let t_poll_done = std::time::Instant::now(); let post_us = t_poll_start.duration_since(t_post_start).as_micros() as u64; @@ -639,24 +666,34 @@ fn serve_get_fallback( return Ok((false, 0, 0, 0, 0)); } let n = segments.len(); + let last_nonempty = segments.iter().rposition(|segment| !segment.is_empty()); let mut offset: u64 = 0; // Hold the MR until poll completes (drop = dereg). - let mut mrs = Vec::with_capacity(n); + let mut mrs: Vec<(MemRegion, Bytes)> = Vec::with_capacity(n); let t_reg_post_start = std::time::Instant::now(); for (i, seg) in segments.iter().enumerate() { if seg.is_empty() { continue; } // RDMA WRITE source side only needs LOCAL access; the LOCAL_WRITE flag matches the slab path convention. - let mr = unsafe { + let mr_result = unsafe { rdma.register_mr_raw( seg.as_ptr() as *mut u8, seg.len(), ibv_access_flags::IBV_ACCESS_LOCAL_WRITE.0, - )? + ) + }; + let mr = match mr_result { + Ok(mr) => mr, + Err(error) => { + for (posted_mr, source) in mrs { + qp.retain_uncertain_write(posted_mr, source); + } + return Err(error); + } }; - let signaled = i + 1 == n; // Only signal on the last one - qp.post_write( + let signaled = Some(i) == last_nonempty; // Last nonempty WRITE signals completion. + if let Err(error) = qp.post_write( i as u64, mr.addr, mr.lkey, @@ -664,13 +701,26 @@ fn serve_get_fallback( dst_rkey, seg.len() as u32, signaled, - )?; + ) { + qp.retain_uncertain_write(mr, seg.clone()); + for (posted_mr, source) in mrs { + qp.retain_uncertain_write(posted_mr, source); + } + return Err(error); + } offset += seg.len() as u64; - mrs.push(mr); + mrs.push((mr, seg.clone())); } let t_poll_start = std::time::Instant::now(); // Wait for the last WRITE to complete (RC guarantees prior ones did too). - RcQp::poll_n(client_cq, 1)?; + if !mrs.is_empty() { + if let Err(error) = RcQp::poll_n_timeout(client_cq, 1, subset_cq_timeout()) { + for (posted_mr, source) in mrs { + qp.retain_uncertain_write(posted_mr, source); + } + return Err(error); + } + } let t_poll_done = std::time::Instant::now(); // At this point mrs drop and dereg. let reg_post_us = t_poll_start.duration_since(t_reg_post_start).as_micros() as u64; @@ -778,6 +828,7 @@ fn try_serve_get_via_slab_with_meta( let t_post_start = std::time::Instant::now(); let mut n_writes_posted = 0u64; let mut outstanding_writes = 0usize; + let mut uncertain_write = false; let mut poll_us = 0u64; let mut had_error: Option = None; let mut first_stream_completion_us: Option = None; @@ -798,6 +849,7 @@ fn try_serve_get_via_slab_with_meta( stripe_len as u32, true, // signaled ) { + uncertain_write = true; had_error = Some(format!( "post RDMA write for stripe {}: {}", stripe_idx, error @@ -807,12 +859,17 @@ fn try_serve_get_via_slab_with_meta( outstanding_writes += 1; if outstanding_writes == RDMA_WRITE_COMPLETION_WINDOW { let poll_start = std::time::Instant::now(); - if let Err(error) = RcQp::poll_n(client_cq, outstanding_writes) { + if let Err(error) = + RcQp::poll_n_timeout(client_cq, outstanding_writes, subset_cq_timeout()) + { + uncertain_write = true; had_error = Some(format!("poll RDMA write completion window: {}", error)); } poll_us += poll_start.elapsed().as_micros() as u64; - outstanding_writes = 0; + if !uncertain_write { + outstanding_writes = 0; + } } } } @@ -833,9 +890,9 @@ fn try_serve_get_via_slab_with_meta( // 5. Drain every posted WRITE before returning, including when a later stripe // failed. The slab extent backs in-flight RNIC DMA and must not be released early. - let poll_result = if outstanding_writes > 0 { + let poll_result = if outstanding_writes > 0 && !uncertain_write { let poll_start = std::time::Instant::now(); - let result = RcQp::poll_n(client_cq, outstanding_writes); + let result = RcQp::poll_n_timeout(client_cq, outstanding_writes, subset_cq_timeout()); poll_us += poll_start.elapsed().as_micros() as u64; result } else { @@ -844,6 +901,19 @@ fn try_serve_get_via_slab_with_meta( let post_us = t_post_done; + if let Err(error) = &poll_result { + uncertain_write = true; + had_error.get_or_insert_with(|| format!("poll final RDMA completions: {error}")); + } + if uncertain_write { + // The extent remains pinned until Drop destroys this QP. The caller must + // not run a fallback GET on the old QP/CQ after a partial/late WRITE. + qp.retain_uncertain_extent(extent); + return Err(anyhow!(RetireLegacyGet( + had_error.unwrap_or_else(|| "RDMA WRITE completion uncertain".to_string()) + ))); + } + if let Some(error) = had_error { return Err(anyhow!("RDMA GET stream failed: {}", error)); } From 66abc66a347225225430a6aab1187cc64d0d5e85 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Fri, 2 Oct 2026 15:06:35 +0800 Subject: [PATCH 18/20] Document multi-rail ownership, upgrade path, and competition evidence Add a reviewable lifecycle and failure design note plus the five-act story, and link both from the main README. Keep Soft-RoCE, Mock, and physical HCA claims separate. --- README.md | 5 ++++ docs/multi-rail-design.md | 58 +++++++++++++++++++++++++++++++++++++++ docs/story.md | 30 ++++++++++++++++++++ 3 files changed, 93 insertions(+) create mode 100644 docs/multi-rail-design.md create mode 100644 docs/story.md diff --git a/README.md b/README.md index aadcdfb..7d9932d 100644 --- a/README.md +++ b/README.md @@ -307,6 +307,11 @@ make bench ### Multi-rail descriptor reads (experimental) +The [five-act competition story](docs/story.md) explains the problem, mechanism +and evidence boundary. The [design and verification note](docs/multi-rail-design.md) +records route semantics, memory ownership, failure behavior, budgets, upgrade +order and measured limits. + The Rust SDK can read one striped object over multiple local RDMA devices and listeners on the same owning storage node. This uses the existing descriptor GET and tag-15 SGE protocol; object placement and disk stripes are unchanged. diff --git a/docs/multi-rail-design.md b/docs/multi-rail-design.md new file mode 100644 index 0000000..f633a07 --- /dev/null +++ b/docs/multi-rail-design.md @@ -0,0 +1,58 @@ +# Multi-rail RDMA read design and verification + +This document describes the opt-in implementation in [draft PR #32](https://github.com/DaoCloud/ContextStore/pull/32), based independently on official `main` at `b5c6451`. It preserves the stored object and disk stripe layout and the existing upper-level read methods. The new Rust SDK entry point is `KvClient::read_multi_rail_into`. + +## Path model and rollout + +A `RailRoute` associates a placement-advertised storage endpoint with a stable rail ID, client local RDMA device/port/GID, remote listener, enabled state and positive capacity weight. Two routes for one owning node must use distinct local ports or devices and distinct listeners. The server's `CS_RDMA_DEVICES` starts a listener and Verbs context for each configured device. The administrator maps both listeners to the same storage owner; the client validates that a route's advertised endpoint matches the placement before it can schedule any stripe. + +Configure and verify the additional server listener first, deploy a client with the new SDK, then enable its second route. A single route uses the same new read state machine. Existing gRPC reads and old RDMA clients continue using their original interfaces. A client that opts into tag-15 scatter reads needs a server version with that wire handler; rolling upgrades must enable multi-rail only after the owning servers have it. There is no automatic listener discovery or protocol negotiation in this version. + +## One read + +```mermaid +flowchart LR + A[LookupObject] --> B[Validate descriptor, placement, coverage and budgets] + B --> C[Assign whole stripes by weighted queued bytes] + C --> R0[Rail 0: QP, CQ, MR, compact buffer] + C --> R1[Rail 1: QP, CQ, MR, compact buffer] + R0 --> J[Join all rail workers] + R1 --> J + J --> V[Check bytes, stripe checksums and post-read lookup] + V -->|all checks pass| P[Publish complete object under cancellation gate] + V -->|failure| F[Release private staging after transport quiescence] +``` + +The planner rejects missing, duplicate, out-of-range or inconsistent stripe descriptions before transport. It keeps each stripe's original object offset and owning endpoint; the assignment changes only which network path carries that stripe. The weighted scheduler favors the rail with less assigned bytes relative to its weight. Each worker registers a compact receive buffer for its own stripes plus the required safe dummy segment, maps the object ranges to tag-15 SGEs, and verifies every SGE stays within its MR. The service writes only the requested stripe subset into those destinations. Slab and registered-buffer fallback paths use the same SGE mapping. + +The reader joins every started worker. It checks the reported byte and chunk counts, requires every planned byte exactly once, and compares per-stripe xxh3 when the placement supplies checksums. A partially populated checksum list is invalid. After transfer, the SDK repeats `LookupObject` and compares object key, handle, Generation, content ETag, Layout Version, size, stripe count/size and complete placement. It copies the assembled object to the caller only if these checks pass. Cancellation and publication hold the same gate; whichever enters first determines whether publication occurs. + +`verify_stripe_checksums` is off by default on the server. A new checksummed test object must be written after enabling it. Old objects without checksums are readable under their original policy, but their reads do not claim per-stripe integrity verification. + +## Ownership and failure semantics + +| Event | Observable result | Lifetime rule | +| --- | --- | --- | +| Route disabled or temporarily cooling down | Another eligible rail may be chosen for a later request; one rail remains valid | No same-request retry after a required rail fails | +| Invalid descriptor/placement or resource limit | Typed error before data transfer | No MR or caller-buffer mutation | +| Rail timeout, disconnect, CQ failure or worker panic | Entire object read fails; no partial publish | Worker stops using QP before deregistering MR and releasing private target memory | +| Checksum mismatch or post-read version/placement change | Entire object read fails | All workers are joined; private data is discarded | +| Cancellation | Error if cancellation wins the publication gate | In-flight rail tasks first quiesce; destination stays unchanged | +| Server WRITE completion uncertain | Server closes that control connection | Source `SlabExtent`, cached pin, or `(MR, Bytes)` remains owned by `RcQp` until its Drop destroys the QP; the old CQ is not reused | +| Client GET control send/receive error | Client QP is destroyed and the control stream shut down | Destination MR may be released or reused only after QP destruction | + +The server's CQ deadline is controlled by `CS_RDMA_CQ_TIMEOUT_MS`, clamped to 100–30,000 ms, defaulting to 30,000 ms. A cache-miss legacy complete-object GET distinguishes a pre-WRITE storage failure, where fallback may still be valid, from an uncertain posted WRITE, where it pins the extent and exits the connection. Cache-hit and per-chunk legacy GETs also retain every posted source on post or CQ errors. This closes the stale-CQE and slab-reuse hole exposed by independent review. The [RXE receipts](../kv-service/benchmarks/results/softroce-vm-test-manifest.json) include a post-handshake 1 ms legacy GET timeout, 2 s server CQ timeout with 0/64 completions, connection retirement and six-second caller-buffer reuse check. + +## Resource bounds and observability + +One `RailReader` reserves active reads, final/per-rail staging bytes, registered lengths, total in-flight object bytes, and per-rail task/in-flight bytes across concurrent calls. Defaults include eight active reads, eight tasks per rail, 4 GiB staging/registration/in-flight ceilings and 2 GiB per-rail in-flight ceiling; callers can lower these in `RailLimits`. On Linux, finite process `RLIMIT_MEMLOCK` caps the effective registration budget at 80% of the soft limit. Preflight reservation returns `ResourceExhausted` before launching excess work. The server QP has a 128-WR send depth and the slab streaming path polls in a bounded completion window. + +`RailSnapshot` and the CLI expose each route's healthy/cooldown status, successful/failed reads, transferred bytes, latency, current and peak in-flight requests/bytes, current and peak registered bytes, local device, listener, and sysfs NUMA/PCI address when present. Multiple separately constructed `RailReader` instances do not share one process-wide budget; deployments using several instances must enforce a shared outer cap. + +Topology is currently reported and selected through explicit route configuration/weights. The scheduler does not yet measure live NIC bandwidth or automatically choose NUMA/PCIe affinity. RXE devices correctly report absent physical PCI/NUMA identity rather than inventing it. + +## Verification and measurement limits + +Run the hardware-independent Rust tests on any Linux host with libibverbs available; the ignored real-Verbs tests require configured listeners. [README](../README.md) lists the commands, the [portable two-VM scripts](../kv-service/deploy/softroce-vm/) recreate two independent RXE paths, and [raw paired samples](../kv-service/benchmarks/results/softroce-vm-paired-samples.csv) plus [analysis](../kv-service/benchmarks/results/softroce-vm-analysis.json) allow recalculation. A [four-minute demonstration](https://github.com/yanchaomei/ContextStore/releases/tag/multi-rail-softroce-demo-2026-10-02) includes the live command/output JSON and an edited, narrated video. + +The paired RXE run uses the same object, layout, virtual disk, guests, server and concurrency within each one/two-rail cell. Single-concurrency ratios at 64/128/256 MiB are 1.071×/1.086×/1.113×; at 64 MiB with four concurrent reads the aggregate ratio is 1.018×. CPU and memory costs rise. These are software-RoCE/QEMU results on one physical host and one virtual disk. A separate physical ConnectX-6 Dx run measures one rail only. Two physical HCA paths, separate NVMe supply, PCIe/NUMA effects and physical link aggregation remain unmeasured; neither the RXE nor Mock results imply those properties. diff --git a/docs/story.md b/docs/story.md new file mode 100644 index 0000000..928ef66 --- /dev/null +++ b/docs/story.md @@ -0,0 +1,30 @@ +# One Worker, two rails, one complete object + +The object stays in its existing disk stripes. A single Worker reads those stripes through two independently controlled RDMA paths and exposes the result only after every required byte, checksum and object-version check succeeds. + +## 1. Conflict + +Concurrent disks can supply a large KV-cache object faster than one network path can carry it. Splitting the transfer across connections is tempting, but direct writes into the caller's buffer make a partial result visible. A timed-out RDMA WRITE or late CQE can outlive the request that owns its destination or source memory. + +## 2. Insight + +The placement already describes the object's physical stripes and owning storage endpoint. It does not require a new disk layout to select a network path. The client can map one advertised endpoint to two listener/device pairs, assign whole stripes by weighted bytes, and receive each rail into private registered memory. The caller's buffer remains untouched until the full object is validated. + +## 3. Mechanism + +`KvClient::read_multi_rail_into` performs lookup and placement validation, asks one shared `RailReader` to plan and transfer the stripes, checks completeness and optional per-stripe xxh3, repeats the lookup to compare descriptor and placement identity, then copies the complete object under the cancellation gate. Each rail owns its QP, CQ and MR. Resource reservations cover both concurrent requests and individual rails. An uncertain WRITE completion retires the connection while its source remains pinned until QP destruction. One configured rail follows the same path and remains supported. + +The transport state machine is deterministic. An LLM or agent is not placed on the latency-sensitive data path because scheduling, bounds and memory ownership need explicit, testable rules. A future operator agent could diagnose metrics or suggest placement changes without controlling the safety path. + +## 4. Evidence + +- [Independent draft PR #32](https://github.com/DaoCloud/ContextStore/pull/32) starts from official `main`, keeps the existing object layout, and contains the implementation, tests and reusable deployment scripts. +- Two isolated KVM guests with two separate tap/bridge/RXE paths completed a 64 MiB / 16-stripe object read over real Verbs; each rail carried 32 MiB. The [four-minute demo and raw live command record](https://github.com/yanchaomei/ContextStore/releases/tag/multi-rail-softroce-demo-2026-10-02) show link shutdown/recovery, corruption/recovery and late second-rail completion. +- The [paired software-RoCE samples](../kv-service/benchmarks/results/softroce-vm-paired-samples.csv) show 1.071×, 1.086× and 1.113× single-to-dual throughput ratios at 64, 128 and 256 MiB with one concurrent read. At 64 MiB and four concurrent reads, the [recorded aggregate ratio](../kv-service/benchmarks/results/softroce-vm-concurrent-64-128-summary.csv) is only 1.018×. CPU cost rises. +- A separate [single-rail physical HCA record](../kv-service/benchmarks/results/2026-10-02-skv-single-hca.json) verifies the Verbs path on ConnectX-6 Dx; it is not a dual-HCA aggregation measurement. + +## 5. Next decision + +Maintainer review should settle whether the explicit listener map belongs in an extended placement protocol and whether checksum verification should become a migration default. Once two independent physical HCA paths and independently measured disk supply are available, measure HCA counters, per-rail utilization, CPU, NUMA/PCIe placement and disk bandwidth under the same single/dual object and concurrency conditions. Those results decide whether the next bottleneck is network, disk, memory or software overhead. + +**Evidence boundary:** Two RXE rails share one host and one virtual disk. They prove real Verbs scheduling and safety behavior, not HCA offload, separate NVMe bandwidth or physical multi-NIC aggregation. Community approval, official CI execution and competition submission are separate gates. From 4e9217d822cba1ce795824ffa55188a26ed03efe Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Sat, 3 Oct 2026 04:02:59 +0800 Subject: [PATCH 19/20] Advertise and discover RDMA rail listeners through placement Add optional fabric/listener capabilities to LookupObject without changing stored stripe layout. Match them to explicit local Verbs paths, reject stale or ambiguous ownership, bound QP tasks, preserve manual routes and old clients, and verify two RXE rails plus rolling-upgrade behavior. --- README.md | 40 +++- docs/multi-rail-design.md | 8 +- docs/rail-capability-discovery.md | 21 ++ docs/story.md | 2 +- .../client-rs/src/bin/rail_read_bench.rs | 77 ++++++- kv-service/client-rs/src/rail_read.rs | 188 +++++++++++++++- kv-service/client-rs/src/rail_read/tests.rs | 207 ++++++++++++++++++ kv-service/client-rs/tests/rail_read_e2e.rs | 102 ++++++++- kv-service/configs/server-softroce-vm.toml | 8 + kv-service/proto/kv_service.proto | 10 + kv-service/server/src/api/service.rs | 178 ++++++++++++++- kv-service/server/src/config.rs | 74 +++++++ 12 files changed, 891 insertions(+), 24 deletions(-) create mode 100644 docs/rail-capability-discovery.md diff --git a/README.md b/README.md index 7d9932d..9157dcc 100644 --- a/README.md +++ b/README.md @@ -315,8 +315,12 @@ order and measured limits. The Rust SDK can read one striped object over multiple local RDMA devices and listeners on the same owning storage node. This uses the existing descriptor GET and tag-15 SGE protocol; object placement and disk stripes are unchanged. -`LookupObject` advertises one primary RDMA endpoint per node, so each rail -explicitly maps that advertised endpoint to a listener on the same node. +`LookupObject` advertises one primary RDMA endpoint per node. New servers can +also attach optional, ephemeral fabric/listener capabilities to Placement; +`RailReader::discover_from_placement` matches those with locally configured +fabric/device/port/GID paths. The server does not need to change or persist +the object's disk stripes. Old clients ignore the extra protobuf field, and +existing explicit `RailRoute` configurations remain usable with old servers. Different rail entries for one node must use different local device/port pairs and different remote listeners. A seventh comma-separated rail field may set a positive relative scheduling weight; omitted weights default to 1. The @@ -324,8 +328,8 @@ reader also reports each local device's sysfs NUMA node and PCI address when the host exposes them. ```bash -# Run against an already stored striped object. Set the server's -# CS_RDMA_DEVICES for both listeners and advertise its primary listener. +# Manual listener mapping against an already stored striped object. +# Set the server's CS_RDMA_DEVICES for both listeners. ./target/release/cs-rail-read-bench \ --environment physical --coordinator http://10.0.0.1:50051 \ --namespace bench --object-key large-object \ @@ -334,7 +338,33 @@ the host exposes them. --warmup 1 --iterations 5 ``` -Use only the first `--rail` for a comparable single-rail run. The CLI labels +For discovered listeners, configure `[[cluster.rdma_rails]]` on the owning +server, as in [the two-RXE example](kv-service/configs/server-softroce-vm.toml). +The client still selects trusted local Verbs devices and fabric IDs, while +`LookupObject` supplies the remote listeners: + +```bash +./target/release/cs-rail-read-bench \ + --environment soft-roce --coordinator http://10.31.0.2:55151 \ + --namespace rust-bench --object-key large-object \ + --local-rail 'r0,fabric-a,rxe_c0,1,1' \ + --local-rail 'r1,fabric-b,rxe_c1,1,1' \ + --warmup 1 --iterations 5 +``` + +The CLI looks up the object before constructing its reader. SDK callers can +call `lookup_object`, then `RailReader::discover_from_placement`, then the +unchanged `read_multi_rail_into` entry point. Discovery rejects an owner or +fabric mismatch; if advertised capabilities change, rebuild the reader. It +does not silently fall back to an arbitrary listener; use +the explicit `--rail` mode for a server without advertised capabilities. +A fixed limit of 32 configured routes caps per-request worker/QP creation +before any RDMA connection is opened, without changing the public `RailLimits` +struct shape. +The [discovery design and upgrade tests](docs/rail-capability-discovery.md) +record the wire contract and one/two-rail compatibility behavior. + +Use only the first `--rail` or `--local-rail` for a comparable single-rail run. The CLI labels physical RDMA and Soft-RoCE separately, reports per-rail bytes, and hashes every returned object. Set the server's cache policy and disk-read forcing identically for both runs; these flags cannot prove a network bottleneck by diff --git a/docs/multi-rail-design.md b/docs/multi-rail-design.md index f633a07..fe4d353 100644 --- a/docs/multi-rail-design.md +++ b/docs/multi-rail-design.md @@ -4,9 +4,9 @@ This document describes the opt-in implementation in [draft PR #32](https://gith ## Path model and rollout -A `RailRoute` associates a placement-advertised storage endpoint with a stable rail ID, client local RDMA device/port/GID, remote listener, enabled state and positive capacity weight. Two routes for one owning node must use distinct local ports or devices and distinct listeners. The server's `CS_RDMA_DEVICES` starts a listener and Verbs context for each configured device. The administrator maps both listeners to the same storage owner; the client validates that a route's advertised endpoint matches the placement before it can schedule any stripe. +A `RailRoute` associates a placement-advertised storage endpoint with a stable rail ID, client local RDMA device/port/GID, remote listener, enabled state and positive capacity weight. Two routes for one owning node must use distinct local ports or devices and distinct listeners. The server's `CS_RDMA_DEVICES` starts a listener and Verbs context for each configured device. Explicit routes remain supported. With optional `[[cluster.rdma_rails]]`, `LookupObject` publishes fabric/listener capabilities only for nodes that own this object's stripes; `LocalRailPath` binds each advertised fabric to a trusted local device/port/GID. The client rejects an owner mismatch, duplicate fabric/listener, or missing local match before transport. The [discovery contract](rail-capability-discovery.md) describes this additive wire field. -Configure and verify the additional server listener first, deploy a client with the new SDK, then enable its second route. A single route uses the same new read state machine. Existing gRPC reads and old RDMA clients continue using their original interfaces. A client that opts into tag-15 scatter reads needs a server version with that wire handler; rolling upgrades must enable multi-rail only after the owning servers have it. There is no automatic listener discovery or protocol negotiation in this version. +Configure and verify the additional server listener first, then advertise its reachable address and fabric ID. Deploy a client with the new SDK and configure matching local fabrics. `discover_from_placement` requires the new optional capabilities; if talking to an older server it returns an explicit error and the existing manual route mode remains available. A single discovered or manual route uses the same read state machine. Existing gRPC reads and old RDMA clients continue using their original interfaces; old protobuf clients ignore the added Placement field. A client that opts into tag-15 scatter reads still needs a server version with that wire handler. Discovery does not infer local GIDs, test reachability in advance, or negotiate protocol versions. ## One read @@ -45,11 +45,11 @@ The server's CQ deadline is controlled by `CS_RDMA_CQ_TIMEOUT_MS`, clamped to 10 ## Resource bounds and observability -One `RailReader` reserves active reads, final/per-rail staging bytes, registered lengths, total in-flight object bytes, and per-rail task/in-flight bytes across concurrent calls. Defaults include eight active reads, eight tasks per rail, 4 GiB staging/registration/in-flight ceilings and 2 GiB per-rail in-flight ceiling; callers can lower these in `RailLimits`. On Linux, finite process `RLIMIT_MEMLOCK` caps the effective registration budget at 80% of the soft limit. Preflight reservation returns `ResourceExhausted` before launching excess work. The server QP has a 128-WR send depth and the slab streaming path polls in a bounded completion window. +One `RailReader` reserves active reads, final/per-rail staging bytes, registered lengths, total in-flight object bytes, and per-rail task/in-flight bytes across concurrent calls. Defaults include eight active reads, eight active tasks per rail, 4 GiB staging/registration/in-flight ceilings and 2 GiB per-rail in-flight ceiling; callers can lower these in `RailLimits`. A fixed source-compatible cap of 32 configured routes bounds worker/QP tasks in one request. On Linux, finite process `RLIMIT_MEMLOCK` caps the effective registration budget at 80% of the soft limit. Route count and preflight reservations return `ResourceExhausted` before launching excess work. The server QP has a 128-WR send depth and the slab streaming path polls in a bounded completion window. `RailSnapshot` and the CLI expose each route's healthy/cooldown status, successful/failed reads, transferred bytes, latency, current and peak in-flight requests/bytes, current and peak registered bytes, local device, listener, and sysfs NUMA/PCI address when present. Multiple separately constructed `RailReader` instances do not share one process-wide budget; deployments using several instances must enforce a shared outer cap. -Topology is currently reported and selected through explicit route configuration/weights. The scheduler does not yet measure live NIC bandwidth or automatically choose NUMA/PCIe affinity. RXE devices correctly report absent physical PCI/NUMA identity rather than inventing it. +Topology is currently reported and selected through local fabric/device configuration and explicit weights. Remote listener discovery does not measure live NIC bandwidth or automatically choose NUMA/PCIe affinity. RXE devices correctly report absent physical PCI/NUMA identity rather than inventing it. ## Verification and measurement limits diff --git a/docs/rail-capability-discovery.md b/docs/rail-capability-discovery.md new file mode 100644 index 0000000..2c373dc --- /dev/null +++ b/docs/rail-capability-discovery.md @@ -0,0 +1,21 @@ +# Optional RDMA rail capability discovery + +## Problem and boundary + +The current `RailReader` safely reads an object over multiple rails, but deployments must manually repeat every remote listener in client configuration. The storage layout and public read method remain unchanged. This extension puts optional network capabilities in the ephemeral `LookupObject` placement response; it does not store them with object metadata or change the stripe owner, offset, checksum or `layout_hash`. + +## Wire and configuration + +`PlacementDescriptor` gains repeated `RdmaRailEndpoint` entries. Each contains the owning node ID, the existing advertised RDMA endpoint used by its chunks, a stable fabric ID, and the actual listener endpoint. Protobuf readers that do not know the new field ignore it. Servers with no `cluster.rdma_rails` configuration emit an empty list and retain their current behavior. The local node config and each optional `cluster.data_nodes` entry may advertise several fabric/listener pairs. Config validation rejects blank IDs/endpoints and duplicates within one node before serving traffic. Bind addresses and advertised listener addresses are intentionally separate; operators must provide reachable addresses. + +## Client mapping + +`LocalRailPath` names one local Verbs device/port/GID and its fabric ID. `RailReader::discover_from_placement` matches an advertised capability to a local path with the same fabric ID, producing ordinary `RailRoute` entries, one per (owning endpoint, fabric). This reuses the existing scheduler, state machine, resource accounting and private receive buffers. Discovery never treats two listeners on the same local port as independent rails or one listener as two storage owners. An internal 32-route cap bounds worker/QP tasks before transfer without changing the public `RailLimits` struct shape; existing nine-route manual configurations remain valid. A missing or ambiguous capability returns an explicit error; callers can still use the existing manual `RailReader::new` path for older servers. It does not attempt to infer GIDs, bypass local device configuration or accept a capability for a different object owner. + +## Consistency and upgrade + +`LookupObject` creates the capability list from current cluster configuration after it knows the actual chunk owners, including the existing `grpc_endpoint` fallback for local or remote `data_nodes` entries configured without an explicit ID. Local ownership follows the same gRPC-address fallback as the existing storage path and also checks the advertised RDMA endpoint. The client validates that every discovered entry matches a chunk owner and its advertised RDMA endpoint, rejects duplicates, and only constructs routes for configured local fabric IDs. A discovered reader pins the capability and owner/endpoint snapshots used to build its routes: if a later `LookupObject` changes either, it rejects the request before transport and must be rebuilt. The existing post-read placement comparison also rejects a change during transfer. Rollout order is server advertisement first, client discovery second; existing clients and manual-route reads remain compatible. + +## Acceptance + +Unit tests cover empty legacy advertisement, one owner with two independent rails, multi-owner routing (including omitted remote node IDs), duplicate/blank configuration, a mismatched owner or endpoint, capability/owner changes after reader construction, cross-owner listener reuse, the worker/QP task cap, and single-rail compatibility. On two isolated RXE paths, the discovery CLI resolved `rxe_c0` to `10.31.0.2:55153` and `rxe_c1` to `10.32.0.2:55154` using only local fabric IDs. One Worker restored the same 64 MiB / 16-stripe object with 32 MiB per rail and xxh3 `a0a4cbfa5cad46af`. An externally disabled second client link left the first rail completed, the second rail failed, and the caller buffer unchanged; after link restoration the same object read succeeded. An old client read the new server's optional field successfully, and a new client used manual routes against the old server. Discovery mode against the old server failed clearly without attempting an arbitrary route. [Raw test receipts](../kv-service/benchmarks/results/softroce-vm-discovery.json) preserve these observations. Physical HCA aggregation is not inferred from this feature. diff --git a/docs/story.md b/docs/story.md index 928ef66..0cac8b1 100644 --- a/docs/story.md +++ b/docs/story.md @@ -8,7 +8,7 @@ Concurrent disks can supply a large KV-cache object faster than one network path ## 2. Insight -The placement already describes the object's physical stripes and owning storage endpoint. It does not require a new disk layout to select a network path. The client can map one advertised endpoint to two listener/device pairs, assign whole stripes by weighted bytes, and receive each rail into private registered memory. The caller's buffer remains untouched until the full object is validated. +The placement already describes the object's physical stripes and owning storage endpoint. It does not require a new disk layout to select a network path. The server can optionally advertise additional fabric/listener capabilities in `LookupObject`; the client matches these to locally configured Verbs devices, assigns whole stripes by weighted bytes, and receives each rail into private registered memory. Older servers remain usable through explicit route mapping. The caller's buffer remains untouched until the full object is validated. ## 3. Mechanism diff --git a/kv-service/client-rs/src/bin/rail_read_bench.rs b/kv-service/client-rs/src/bin/rail_read_bench.rs index 29a8e4e..30b2a49 100644 --- a/kv-service/client-rs/src/bin/rail_read_bench.rs +++ b/kv-service/client-rs/src/bin/rail_read_bench.rs @@ -2,7 +2,7 @@ use anyhow::{anyhow, Context, Result}; use clap::Parser; -use contextstore_client_rs::rail_read::{RailLimits, RailReader, RailRoute}; +use contextstore_client_rs::rail_read::{LocalRailPath, RailLimits, RailReader, RailRoute}; use contextstore_client_rs::rdma::RdmaClientConfig; use contextstore_client_rs::KvClient; use std::sync::Arc; @@ -18,8 +18,12 @@ struct Args { #[arg(long)] object_key: String, /// Repeat: id,local_device,advertised_endpoint,listener,port,gid_index[,weight]. - #[arg(long = "rail", required = true)] + #[arg(long = "rail", conflicts_with = "local_rails")] rails: Vec, + /// Repeat: id,fabric_id,local_device,port,gid_index[,weight]. + /// Remote listener is discovered from the object's PlacementDescriptor. + #[arg(long = "local-rail", conflicts_with = "rails")] + local_rails: Vec, /// State which real Verbs environment produced these measurements. #[arg(long, value_parser = ["physical", "soft-roce"])] environment: String, @@ -61,6 +65,40 @@ fn parse_route(spec: &str) -> Result { .with_weight(weight)) } +fn parse_local_path(spec: &str) -> Result { + let fields: Vec<_> = spec.split(',').map(str::trim).collect(); + if !(5..=6).contains(&fields.len()) || fields[..3].iter().any(|field| field.is_empty()) { + return Err(anyhow!( + "local rail spec must be id,fabric_id,device,port,gid_index[,weight]" + )); + } + let port = fields[3] + .parse::() + .context("local rail port is not a u8")?; + if port == 0 { + return Err(anyhow!("local rail port must be positive")); + } + let gid = fields[4] + .parse::() + .context("local rail GID index is not a u8")?; + let weight = fields + .get(5) + .map(|value| { + value + .parse::() + .context("local rail weight is not a u32") + }) + .transpose()? + .unwrap_or(1); + if weight == 0 { + return Err(anyhow!("local rail weight must be positive")); + } + Ok(LocalRailPath::new(fields[0], fields[1], fields[2]) + .with_port(port) + .with_gid_index(gid) + .with_weight(weight)) +} + #[cfg(target_os = "linux")] fn process_usage() -> (u64, u64, u64) { let mut usage = unsafe { std::mem::zeroed::() }; @@ -205,12 +243,21 @@ async fn main() -> Result<()> { "--iterations must be positive and --concurrency must be 1..=8" )); } - let routes = args + if args.rails.is_empty() == args.local_rails.is_empty() { + return Err(anyhow!( + "provide either --rail for explicit listeners or --local-rail for discovered listeners" + )); + } + let explicit_routes = args .rails .iter() .map(|spec| parse_route(spec)) .collect::>>()?; - let reader = Arc::new(RailReader::new(routes.clone(), RailLimits::default())?); + let local_paths = args + .local_rails + .iter() + .map(|spec| parse_local_path(spec)) + .collect::>>()?; let endpoint = if args.coordinator.starts_with("http://") || args.coordinator.starts_with("https://") { args.coordinator.clone() @@ -224,9 +271,18 @@ async fn main() -> Result<()> { .lookup_object(&args.namespace, &args.object_key) .await? .ok_or_else(|| anyhow!("object not found"))?; + let reader = Arc::new(if local_paths.is_empty() { + RailReader::new(explicit_routes, RailLimits::default())? + } else { + let placement = lookup + .placement + .as_ref() + .ok_or_else(|| anyhow!("object lookup returned no placement"))?; + RailReader::discover_from_placement(placement, &local_paths, RailLimits::default())? + }); let size = usize::try_from(lookup.descriptor.size)?; let mut destination = vec![0u8; size]; - for route in &routes { + for route in reader.routes() { println!( "route,id={},device={},port={},gid={},weight={},advertised={},listener={}", route.id, @@ -283,7 +339,7 @@ async fn main() -> Result<()> { println!( "sample,environment={},rails={},iteration={},bytes={},latency_us={},gib_per_s={gib_per_s:.3},xxh3={checksum:016x}", args.environment, - routes.len(), + reader.routes().len(), iteration + 1, bytes, elapsed.as_micros(), @@ -301,7 +357,7 @@ async fn main() -> Result<()> { println!( "summary,environment={},rails={},bytes_per_iter={},iters={},median_us={},cpu_user_us={},cpu_system_us={},peak_rss_kb={},rail_bytes={rail_bytes:?}", args.environment, - routes.len(), + reader.routes().len(), size, args.iterations, median.as_micros(), @@ -355,4 +411,11 @@ mod tests { let route = parse_route("r1,mlx5_1,10.0.0.1:50053,10.0.1.1:50054,1,3,4").unwrap(); assert_eq!(route.weight, 4); } + + #[test] + fn local_rail_spec_names_a_fabric_without_hardcoding_a_listener() { + assert!(parse_local_path("r1,fabric-b,mlx5_1,1,3,4").is_ok()); + assert!(parse_local_path("r1,,mlx5_1,1,3").is_err()); + assert!(parse_local_path("r1,fabric-b,mlx5_1,0,3").is_err()); + } } diff --git a/kv-service/client-rs/src/rail_read.rs b/kv-service/client-rs/src/rail_read.rs index 4ae199c..f8490a7 100644 --- a/kv-service/client-rs/src/rail_read.rs +++ b/kv-service/client-rs/src/rail_read.rs @@ -2,7 +2,7 @@ use crate::pb; use crate::rdma::{RdmaClient, RdmaClientConfig, RdmaReadOutcome}; -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::fmt; use std::path::Path; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; @@ -86,6 +86,50 @@ impl RailRoute { } } +/// A local Verbs path whose fabric ID is matched to a server-advertised listener. +/// No remote endpoint is accepted from local configuration in discovery mode. +#[derive(Clone, Debug)] +pub struct LocalRailPath { + id: String, + fabric_id: String, + device: String, + port: u8, + gid_index: u8, + weight: u32, +} + +impl LocalRailPath { + pub fn new( + id: impl Into, + fabric_id: impl Into, + device: impl Into, + ) -> Self { + Self { + id: id.into(), + fabric_id: fabric_id.into(), + device: device.into(), + port: 1, + gid_index: 3, + weight: 1, + } + } + + pub fn with_port(mut self, port: u8) -> Self { + self.port = port; + self + } + + pub fn with_gid_index(mut self, gid_index: u8) -> Self { + self.gid_index = gid_index; + self + } + + pub fn with_weight(mut self, weight: u32) -> Self { + self.weight = weight.max(1); + self + } +} + /// Bounds all allocations and transfers made by one reader instance. #[derive(Clone, Debug)] pub struct RailLimits { @@ -534,9 +578,14 @@ struct BudgetState { rail_inflight_bytes: Vec, } +// Preserve the RailLimits struct API while bounding one request's worker/QP count. +const MAX_RAIL_TASKS_PER_READ: usize = 32; + /// Bounded multi-rail reader; one active request uses one QP per chosen rail. pub struct RailReader { routes: Vec, + discovered_capabilities: Option>, + discovered_owners: Option>, topologies: Vec, limits: RailLimits, counters: Vec, @@ -619,6 +668,105 @@ fn process_memlock_limit() -> Option { } impl RailReader { + /// Resolve remote listeners advertised by `LookupObject` against known + /// local fabrics. Existing manual routes remain available for older servers. + pub fn discover_from_placement( + placement: &pb::PlacementDescriptor, + local_paths: &[LocalRailPath], + limits: RailLimits, + ) -> Result { + if placement.chunks.is_empty() || placement.rdma_rails.is_empty() { + return Err(RailReadError::InvalidPlacement( + "placement has no rail capabilities; use manual routes for older servers".into(), + )); + } + let mut local_ids = HashSet::new(); + let mut local_fabrics = HashSet::new(); + let mut local_ports = HashSet::new(); + for path in local_paths { + if path.id.trim().is_empty() + || path.fabric_id.trim().is_empty() + || path.device.trim().is_empty() + || path.port == 0 + || !local_ids.insert(path.id.as_str()) + || !local_fabrics.insert(path.fabric_id.as_str()) + || !local_ports.insert((path.device.as_str(), path.port)) + { + return Err(RailReadError::InvalidPlacement( + "local rail IDs, fabrics, devices and ports must be distinct".into(), + )); + } + } + let mut owners = HashSet::new(); + let mut endpoint_owners = HashMap::new(); + for chunk in &placement.chunks { + if chunk.node_id.trim().is_empty() || chunk.rdma_endpoint.trim().is_empty() { + return Err(RailReadError::InvalidPlacement( + "stripe owner or RDMA endpoint is empty".into(), + )); + } + if let Some(previous) = endpoint_owners.insert(&chunk.rdma_endpoint, &chunk.node_id) { + if previous != &chunk.node_id { + return Err(RailReadError::InvalidPlacement( + "one RDMA endpoint identifies two storage owners".into(), + )); + } + } + owners.insert((&chunk.node_id, &chunk.rdma_endpoint)); + } + let mut seen_fabrics = HashSet::new(); + let mut seen_listeners = HashSet::new(); + let mut routes = Vec::new(); + for rail in &placement.rdma_rails { + if rail.node_id.trim().is_empty() + || rail.advertised_endpoint.trim().is_empty() + || rail.fabric_id.trim().is_empty() + || rail.listener_endpoint.trim().is_empty() + || !owners.contains(&(&rail.node_id, &rail.advertised_endpoint)) + || !seen_fabrics.insert((&rail.node_id, &rail.advertised_endpoint, &rail.fabric_id)) + || !seen_listeners.insert(&rail.listener_endpoint) + { + return Err(RailReadError::InvalidPlacement( + "advertised rail is blank, duplicated, or not an object owner".into(), + )); + } + if let Some(path) = local_paths + .iter() + .find(|path| path.fabric_id == rail.fabric_id) + { + let connection = RdmaClientConfig::new(&rail.listener_endpoint, &path.device) + .with_port(path.port) + .with_gid_index(path.gid_index); + routes.push( + RailRoute::new( + format!("{}/{}/{}", rail.node_id, rail.advertised_endpoint, path.id), + &rail.advertised_endpoint, + connection, + ) + .with_weight(path.weight), + ); + } + } + if owners.iter().any(|(_, endpoint)| { + !routes + .iter() + .any(|route| route.advertised_endpoint == **endpoint) + }) { + return Err(RailReadError::InvalidPlacement( + "no matching local fabric for a storage owner".into(), + )); + } + let mut reader = Self::new(routes, limits)?; + reader.discovered_capabilities = Some(placement.rdma_rails.clone()); + reader.discovered_owners = Some( + owners + .into_iter() + .map(|(node, endpoint)| (node.clone(), endpoint.clone())) + .collect(), + ); + Ok(reader) + } + /// Validate routes and create an RDMA reader without opening connections. pub fn new(routes: Vec, mut limits: RailLimits) -> Result { if routes.is_empty() @@ -629,6 +777,12 @@ impl RailReader { "at least one rail and one active read slot are required".into(), )); } + if routes.len() > MAX_RAIL_TASKS_PER_READ { + return Err(RailReadError::ResourceExhausted(format!( + "configured rail count exceeds per-read task limit ({})", + MAX_RAIL_TASKS_PER_READ + ))); + } let mut ids = HashSet::new(); let mut paths = HashSet::new(); let mut local_ports = HashSet::new(); @@ -671,6 +825,8 @@ impl RailReader { effective_registered_budget(limits.max_registered_bytes, process_memlock_limit()); Ok(Self { routes, + discovered_capabilities: None, + discovered_owners: None, topologies, limits, counters, @@ -682,6 +838,11 @@ impl RailReader { }) } + /// The resolved route map used by this reader, including discovered listeners. + pub fn routes(&self) -> &[RailRoute] { + &self.routes + } + /// Read per-rail counters without blocking in-flight transfers. pub fn snapshots(&self) -> Vec { self.routes @@ -788,6 +949,31 @@ impl RailReader { transport: &T, cancel: Option<&RailCancel>, ) -> Result { + if let Some(expected) = &self.discovered_owners { + let current: HashSet<(&str, &str)> = placement + .chunks + .iter() + .map(|chunk| (chunk.node_id.as_str(), chunk.rdma_endpoint.as_str())) + .collect(); + if current.len() != expected.len() + || expected + .iter() + .any(|(node, endpoint)| !current.contains(&(node.as_str(), endpoint.as_str()))) + { + return Err(RailReadError::InvalidPlacement( + "stripe owner changed; rebuild the discovered reader".into(), + )); + } + } + if self + .discovered_capabilities + .as_ref() + .is_some_and(|expected| expected != &placement.rdma_rails) + { + return Err(RailReadError::InvalidPlacement( + "advertised rail capabilities changed; rebuild the discovered reader".into(), + )); + } let mut available = self.routes.clone(); for (route, counters) in available.iter_mut().zip(&self.counters) { route.enabled = counters.is_available(); diff --git a/kv-service/client-rs/src/rail_read/tests.rs b/kv-service/client-rs/src/rail_read/tests.rs index 8aaab75..0558bb1 100644 --- a/kv-service/client-rs/src/rail_read/tests.rs +++ b/kv-service/client-rs/src/rail_read/tests.rs @@ -121,6 +121,213 @@ fn two_independent_listeners_restore_one_unmodified_placement() { assert!(snapshots.iter().all(|rail| rail.peak_registered_bytes > 0)); } +#[test] +fn advertised_rails_resolve_local_fabrics_and_restore_one_object() { + let bytes: Vec = (0..64).map(|index| index as u8).collect(); + let mock = MockTransport::new(bytes.clone()); + let (descriptor, mut placement) = fixture(64, 8); + placement.rdma_rails = vec![ + pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }, + pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-b".into(), + listener_endpoint: "10.0.1.1:50054".into(), + }, + ]; + let local = vec![ + LocalRailPath::new("rail0", "fabric-a", "mock0"), + LocalRailPath::new("rail1", "fabric-b", "mock1"), + ]; + let reader = RailReader::discover_from_placement(&placement, &local, RailLimits::default()) + .expect("two discovered rails"); + assert_eq!(reader.snapshots().len(), 2); + assert_eq!(reader.snapshots()[1].listener, "10.0.1.1:50054"); + let mut destination = vec![0xA5; 64]; + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + assert_eq!(destination, bytes); + assert_eq!(reader.snapshots()[0].bytes, 32); + assert_eq!(reader.snapshots()[1].bytes, 32); +} + +#[test] +fn discovery_rejects_a_capability_for_another_owner() { + let (_descriptor, mut placement) = fixture(64, 8); + placement.rdma_rails = vec![pb::RdmaRailEndpoint { + node_id: "unexpected-owner".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }]; + assert!(RailReader::discover_from_placement( + &placement, + &[LocalRailPath::new("rail0", "fabric-a", "mock0")], + RailLimits::default(), + ) + .is_err()); +} + +#[test] +fn discovery_maps_one_local_fabric_to_each_actual_storage_owner() { + let bytes: Vec = (0..16).map(|index| index as u8).collect(); + let mock = MockTransport::new(bytes.clone()); + let (descriptor, mut placement) = fixture(16, 8); + placement.chunks[1].node_id = "node-b".into(); + placement.chunks[1].rdma_endpoint = "10.0.0.2:50053".into(); + placement.rdma_rails = vec![ + pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }, + pb::RdmaRailEndpoint { + node_id: "node-b".into(), + advertised_endpoint: "10.0.0.2:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.2:50053".into(), + }, + ]; + let reader = RailReader::discover_from_placement( + &placement, + &[LocalRailPath::new("rail0", "fabric-a", "mock0")], + RailLimits::default(), + ) + .unwrap(); + assert_eq!(reader.routes().len(), 2); + let mut destination = vec![0xA5; 16]; + reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .unwrap(); + assert_eq!(destination, bytes); + let calls = mock.calls.lock().unwrap(); + assert!(calls.contains(&"10.0.0.1:50053".to_string())); + assert!(calls.contains(&"10.0.0.2:50053".to_string())); +} + +#[test] +fn discovery_rejects_duplicate_fabrics_and_legacy_absence() { + let (_descriptor, mut placement) = fixture(16, 8); + let local = [LocalRailPath::new("rail0", "fabric-a", "mock0")]; + assert!( + RailReader::discover_from_placement(&placement, &local, RailLimits::default()).is_err() + ); + let endpoint = pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }; + placement.rdma_rails = vec![endpoint.clone(), endpoint]; + assert!( + RailReader::discover_from_placement(&placement, &local, RailLimits::default()).is_err() + ); +} + +#[test] +fn discovered_reader_rejects_a_changed_listener_before_transport() { + let (descriptor, mut placement) = fixture(16, 8); + placement.rdma_rails = vec![pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }]; + let reader = RailReader::discover_from_placement( + &placement, + &[LocalRailPath::new("rail0", "fabric-a", "mock0")], + RailLimits::default(), + ) + .unwrap(); + placement.rdma_rails[0].listener_endpoint = "10.0.1.1:50054".into(); + let mock = MockTransport::new(vec![0x42; 16]); + let mut destination = vec![0xA5; 16]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + assert!(mock.calls.lock().unwrap().is_empty()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} + +#[test] +fn discovered_reader_rejects_a_changed_storage_owner_before_transport() { + let (descriptor, mut placement) = fixture(16, 8); + placement.rdma_rails = vec![pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }]; + let reader = RailReader::discover_from_placement( + &placement, + &[LocalRailPath::new("rail0", "fabric-a", "mock0")], + RailLimits::default(), + ) + .unwrap(); + placement.chunks[1].node_id = "unexpected-owner".into(); + let mock = MockTransport::new(vec![0x42; 16]); + let mut destination = vec![0xA5; 16]; + assert!(reader + .read_into_with(&descriptor, &placement, &mut destination, &mock, None) + .is_err()); + assert!(mock.calls.lock().unwrap().is_empty()); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} + +#[test] +fn discovery_rejects_one_listener_claimed_by_two_owners() { + let (_descriptor, mut placement) = fixture(16, 8); + placement.chunks[1].node_id = "node-b".into(); + placement.chunks[1].rdma_endpoint = "10.0.0.2:50053".into(); + placement.rdma_rails = vec![ + pb::RdmaRailEndpoint { + node_id: "node-a".into(), + advertised_endpoint: "10.0.0.1:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.9:50054".into(), + }, + pb::RdmaRailEndpoint { + node_id: "node-b".into(), + advertised_endpoint: "10.0.0.2:50053".into(), + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.9:50054".into(), + }, + ]; + assert!(RailReader::discover_from_placement( + &placement, + &[LocalRailPath::new("rail0", "fabric-a", "mock0")], + RailLimits::default(), + ) + .is_err()); +} + +#[test] +fn fixed_task_cap_preserves_nine_routes_and_rejects_excess() { + let make_routes = |count| { + (0..count) + .map(|index| { + RailRoute::new( + format!("rail{index}"), + "10.0.0.1:50053", + crate::rdma::RdmaClientConfig::new( + format!("10.0.0.1:{}", 50053 + index), + format!("mock{index}"), + ), + ) + }) + .collect() + }; + assert!(RailReader::new(make_routes(9), RailLimits::default()).is_ok()); + assert!(RailReader::new(make_routes(33), RailLimits::default()).is_err()); +} + #[test] fn partial_rail_failure_leaves_caller_buffer_unchanged() { let mut mock = MockTransport::new(vec![0x42; 64]); diff --git a/kv-service/client-rs/tests/rail_read_e2e.rs b/kv-service/client-rs/tests/rail_read_e2e.rs index a261efe..015b6cc 100644 --- a/kv-service/client-rs/tests/rail_read_e2e.rs +++ b/kv-service/client-rs/tests/rail_read_e2e.rs @@ -5,7 +5,9 @@ #![cfg(feature = "rdma")] -use contextstore_client_rs::rail_read::{RailCancel, RailLimits, RailReader, RailRoute}; +use contextstore_client_rs::rail_read::{ + LocalRailPath, RailCancel, RailLimits, RailReader, RailRoute, +}; use contextstore_client_rs::rdma::{RdmaClient, RdmaClientConfig}; use contextstore_client_rs::KvClient; use prost::bytes::Bytes; @@ -526,6 +528,104 @@ async fn same_object_matches_single_and_dual_rail() { assert_eq!(single_bytes, dual_bytes); } +fn discovered_local_paths() -> [LocalRailPath; 2] { + [ + LocalRailPath::new( + "rail0", + setting("CS_RAIL_FABRIC0", "fabric-a"), + setting("CS_RAIL_DEVICE0", "rxe_c0"), + ) + .with_gid_index( + setting("CS_RAIL_GID0", "1") + .parse::() + .expect("first rail GID"), + ), + LocalRailPath::new( + "rail1", + setting("CS_RAIL_FABRIC1", "fabric-b"), + setting("CS_RAIL_DEVICE1", "rxe_c1"), + ) + .with_gid_index( + setting("CS_RAIL_GID1", "1") + .parse::() + .expect("second rail GID"), + ), + ] +} + +#[tokio::test] +#[ignore = "requires two real rails and server-advertised fabric/listener capabilities"] +async fn advertised_two_rails_restore_one_object() { + let (mut client, key, payload, _advertised) = seeded_object().await; + let lookup = client + .lookup_object("rail-e2e", &key) + .await + .expect("lookup") + .expect("seeded object"); + let reader = Arc::new( + RailReader::discover_from_placement( + lookup.placement.as_ref().expect("placement"), + &discovered_local_paths(), + RailLimits::default(), + ) + .expect("discover rail listeners"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert_eq!( + client + .read_multi_rail_into( + Arc::clone(&reader), + "rail-e2e", + &key, + &mut destination, + None + ) + .await + .expect("discovered dual read"), + Some(payload.len()) + ); + assert_eq!(destination, payload); + assert_eq!(reader.snapshots().len(), 2); + assert!(reader.snapshots().iter().all(|rail| rail.bytes > 0)); +} + +#[tokio::test] +#[ignore = "requires advertised rails; externally disable the second client RXE link"] +async fn advertised_second_link_down_cannot_publish_partial_object() { + let (mut client, key, payload, _advertised) = seeded_object().await; + let lookup = client + .lookup_object("rail-e2e", &key) + .await + .expect("lookup") + .expect("seeded object"); + let reader = Arc::new( + RailReader::discover_from_placement( + lookup.placement.as_ref().expect("placement"), + &discovered_local_paths(), + RailLimits { + io_timeout: Duration::from_secs(4), + ..RailLimits::default() + }, + ) + .expect("discover rail listeners"), + ); + let mut destination = vec![0xA5; payload.len()]; + assert!(client + .read_multi_rail_into( + Arc::clone(&reader), + "rail-e2e", + &key, + &mut destination, + None + ) + .await + .is_err()); + let rails = reader.snapshots(); + assert_eq!(rails[0].reads_ok, 1, "first rail should have completed"); + assert_eq!(rails[1].reads_err, 1, "disabled second rail should fail"); + assert!(destination.iter().all(|byte| *byte == 0xA5)); +} + #[tokio::test] #[ignore = "requires one reachable RDMA listener and one injected dead listener"] async fn failed_second_rail_does_not_publish_partial_bytes() { diff --git a/kv-service/configs/server-softroce-vm.toml b/kv-service/configs/server-softroce-vm.toml index 6aaafd4..6eee800 100644 --- a/kv-service/configs/server-softroce-vm.toml +++ b/kv-service/configs/server-softroce-vm.toml @@ -8,6 +8,14 @@ grpc_advertise = "10.31.0.2:55151" rdma_advertise = "10.31.0.2:55153" data_nodes = [] +[[cluster.rdma_rails]] +fabric_id = "fabric-a" +listener_endpoint = "10.31.0.2:55153" + +[[cluster.rdma_rails]] +fabric_id = "fabric-b" +listener_endpoint = "10.32.0.2:55154" + [storage] devices = ["/home/railtest/data/rail0", "/home/railtest/data/rail1"] data_subdir = "contextstore" diff --git a/kv-service/proto/kv_service.proto b/kv-service/proto/kv_service.proto index 58581a2..b1461c5 100644 --- a/kv-service/proto/kv_service.proto +++ b/kv-service/proto/kv_service.proto @@ -96,6 +96,15 @@ message PlacementChunk { string checksum = 9; // Optional xxh3-64 checksum, set when stripe integrity is enabled } +// Optional, ephemeral network capabilities for a node already owning stripes. +// Old clients ignore this field; it does not alter the stored stripe layout. +message RdmaRailEndpoint { + string node_id = 1; + string advertised_endpoint = 2; // The existing PlacementChunk.rdma_endpoint + string fabric_id = 3; // Matches a configured local Verbs path + string listener_endpoint = 4; // Reachable control listener for this fabric +} + message PlacementDescriptor { ObjectKey key = 1; uint64 placement_epoch = 2; // Current cluster placement-rule/topology version @@ -105,6 +114,7 @@ message PlacementDescriptor { string primary_grpc_endpoint = 6; string primary_rdma_endpoint = 7; repeated PlacementChunk chunks = 8; + repeated RdmaRailEndpoint rdma_rails = 9; } enum CompressionType { diff --git a/kv-service/server/src/api/service.rs b/kv-service/server/src/api/service.rs index 94c6de7..d5b115a 100644 --- a/kv-service/server/src/api/service.rs +++ b/kv-service/server/src/api/service.rs @@ -1,5 +1,6 @@ //! gRPC service handler implementation +use std::collections::HashSet; use std::pin::Pin; use std::sync::Arc; use std::time::{Duration, Instant}; @@ -105,6 +106,7 @@ mod tests { node_id: id.to_string(), grpc_endpoint: grpc.to_string(), rdma_endpoint: String::new(), + rdma_rails: Vec::new(), } } @@ -157,6 +159,114 @@ mod tests { validate_descriptor(&desc, &meta).unwrap(); } + #[test] + fn lookup_placement_advertises_two_local_rails_without_changing_stripes() { + let mut config = crate::config::Config::default(); + config.cluster.node_id = "owner-a".into(); + config.cluster.grpc_advertise = "10.31.0.2:55151".into(); + config.cluster.rdma_advertise = "10.31.0.2:55153".into(); + config.cluster.rdma_rails = vec![ + crate::config::RdmaRailAdvertiseConfig { + fabric_id: "rail-a".into(), + listener_endpoint: "10.31.0.2:55153".into(), + }, + crate::config::RdmaRailAdvertiseConfig { + fabric_id: "rail-b".into(), + listener_endpoint: "10.32.0.2:55154".into(), + }, + ]; + config.metadata.redis_url = "memory://local-rail-advertisement".into(); + let mut legacy_config = config.clone(); + legacy_config.cluster.rdma_rails.clear(); + legacy_config.metadata.redis_url = "memory://local-rail-advertisement-legacy".into(); + let ctx = KVServiceContext::new(config).unwrap(); + let placement = placement_from_meta(&ctx, &key(), &meta()); + let legacy_placement = placement_from_meta( + &KVServiceContext::new(legacy_config).unwrap(), + &key(), + &meta(), + ); + + assert_eq!(placement.chunks.len(), 1); + assert_eq!(placement.chunks, legacy_placement.chunks); + assert_eq!(placement.layout_hash, legacy_placement.layout_hash); + assert_eq!(placement.placement_epoch, legacy_placement.placement_epoch); + assert_eq!(placement.chunks[0].rdma_endpoint, "10.31.0.2:55153"); + assert_eq!(placement.rdma_rails.len(), 2); + assert_eq!(placement.rdma_rails[0].node_id, "owner-a"); + assert_eq!( + placement.rdma_rails[0].advertised_endpoint, + "10.31.0.2:55153" + ); + assert_eq!(placement.rdma_rails[1].fabric_id, "rail-b"); + assert_eq!(placement.rdma_rails[1].listener_endpoint, "10.32.0.2:55154"); + } + + #[test] + fn empty_rail_advertisement_keeps_legacy_placement() { + let ctx = ctx_with_nodes(vec![]); + let placement = placement_from_meta(&ctx, &key(), &meta()); + assert!(placement.rdma_rails.is_empty()); + } + + #[test] + fn rail_capabilities_follow_actual_local_and_remote_owners() { + let mut remote = data_node("owner-b", "10.0.0.2:50051"); + // Config permits an empty ID and uses the gRPC endpoint as the node ID. + remote.node_id.clear(); + remote.rdma_endpoint = "10.0.0.2:50053".into(); + remote.rdma_rails = vec![crate::config::RdmaRailAdvertiseConfig { + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.2:50053".into(), + }]; + let mut config = crate::config::Config::default(); + config.cluster.node_id = "owner-a".into(); + config.cluster.rdma_advertise = "10.0.0.1:50053".into(); + config.cluster.rdma_rails = vec![crate::config::RdmaRailAdvertiseConfig { + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }]; + config.cluster.data_nodes = vec![remote]; + config.metadata.redis_url = "memory://multi-owner-rail-advertisement".into(); + let ctx = KVServiceContext::new(config).unwrap(); + let mut chunks = placement_from_meta(&ctx, &key(), &meta()).chunks; + let mut remote_chunk = chunks[0].clone(); + remote_chunk.node_id = "10.0.0.2:50051".into(); + remote_chunk.rdma_endpoint = "10.0.0.2:50053".into(); + chunks.push(remote_chunk); + + let rails = advertised_rails_for_chunks(&ctx, &chunks); + assert_eq!(rails.len(), 2); + assert_eq!(rails[0].node_id, "owner-a"); + assert_eq!(rails[1].node_id, "10.0.0.2:50051"); + assert_eq!(rails[1].listener_endpoint, "10.0.0.2:50053"); + } + + #[test] + fn local_cluster_entry_without_node_id_uses_local_rail_advertisements() { + let mut config = crate::config::Config::default(); + config.cluster.node_id = "owner-a".into(); + config.cluster.grpc_advertise = "10.0.0.1:50051".into(); + config.cluster.rdma_advertise = "10.0.0.1:50053".into(); + config.cluster.rdma_rails = vec![crate::config::RdmaRailAdvertiseConfig { + fabric_id: "fabric-a".into(), + listener_endpoint: "10.0.0.1:50053".into(), + }]; + config.cluster.data_nodes = vec![ClusterNodeConfig { + node_id: String::new(), + grpc_endpoint: "10.0.0.1:50051".into(), + rdma_endpoint: "10.0.0.1:50053".into(), + rdma_rails: Vec::new(), + }]; + config.metadata.redis_url = "memory://local-node-id-fallback-rails".into(); + let ctx = KVServiceContext::new(config).unwrap(); + let mut chunk = placement_from_meta(&ctx, &key(), &meta()).chunks.remove(0); + chunk.node_id = "10.0.0.1:50051".into(); + let rails = advertised_rails_for_chunks(&ctx, &[chunk]); + assert_eq!(rails.len(), 1); + assert_eq!(rails[0].node_id, "10.0.0.1:50051"); + } + #[test] fn descriptor_validation_rejects_stale_generation() { let meta = meta(); @@ -1534,17 +1644,21 @@ fn configured_data_nodes(ctx: &KVServiceContext) -> Vec { .data_nodes .iter() .map(|n: &ClusterNodeConfig| DataNode { - node_id: if n.node_id.is_empty() { - n.grpc_endpoint.clone() - } else { - n.node_id.clone() - }, + node_id: configured_node_id(n).to_string(), grpc_endpoint: n.grpc_endpoint.clone(), rdma_endpoint: n.rdma_endpoint.clone(), }) .collect() } +fn configured_node_id(node: &ClusterNodeConfig) -> &str { + if node.node_id.is_empty() { + &node.grpc_endpoint + } else { + &node.node_id + } +} + fn is_local_node(ctx: &KVServiceContext, node: &DataNode) -> bool { let local = local_node(ctx); node.node_id == local.node_id || node.grpc_endpoint == local.grpc_endpoint @@ -1741,6 +1855,7 @@ fn placement_from_meta( )); } + let rdma_rails = advertised_rails_for_chunks(ctx, &chunks); pb::PlacementDescriptor { key: Some(internal_key_to_pb(key)), placement_epoch, @@ -1750,7 +1865,59 @@ fn placement_from_meta( primary_grpc_endpoint: local.grpc_endpoint, primary_rdma_endpoint: local.rdma_endpoint, chunks, + rdma_rails, + } +} + +fn advertised_rails_for_chunks( + ctx: &KVServiceContext, + chunks: &[pb::PlacementChunk], +) -> Vec { + if ctx.config.cluster.rdma_rails.is_empty() + && ctx + .config + .cluster + .data_nodes + .iter() + .all(|node| node.rdma_rails.is_empty()) + { + return Vec::new(); + } + let local = local_node(ctx); + let mut seen_owners = HashSet::new(); + let mut advertised = Vec::new(); + for chunk in chunks { + if chunk.rdma_endpoint.is_empty() + || !seen_owners.insert((chunk.node_id.clone(), chunk.rdma_endpoint.clone())) + { + continue; + } + let configured = if (chunk.node_id == local.node_id + || chunk.grpc_endpoint == local.grpc_endpoint) + && chunk.rdma_endpoint == local.rdma_endpoint + { + Some(ctx.config.cluster.rdma_rails.as_slice()) + } else { + ctx.config + .cluster + .data_nodes + .iter() + .find(|node| { + configured_node_id(node) == chunk.node_id + && node.rdma_endpoint == chunk.rdma_endpoint + }) + .map(|node| node.rdma_rails.as_slice()) + }; + if let Some(rails) = configured { + advertised.extend(rails.iter().map(|rail| pb::RdmaRailEndpoint { + node_id: chunk.node_id.clone(), + advertised_endpoint: chunk.rdma_endpoint.clone(), + fabric_id: rail.fabric_id.clone(), + listener_endpoint: rail.listener_endpoint.clone(), + })); + } } + advertised } fn key_from_descriptor(desc: &pb::ObjectDescriptor) -> Result { @@ -2202,6 +2369,7 @@ impl pb::kv_service_server::KvService for KVServiceImpl { primary_grpc_endpoint: local.grpc_endpoint, primary_rdma_endpoint: local.rdma_endpoint, chunks, + rdma_rails: Vec::new(), }; Ok(Response::new(pb::PrepareDistributedPutResponse { accepted: true, diff --git a/kv-service/server/src/config.rs b/kv-service/server/src/config.rs index 096d1c9..e8d3a0d 100644 --- a/kv-service/server/src/config.rs +++ b/kv-service/server/src/config.rs @@ -3,6 +3,7 @@ //! Corresponds to configs/server.toml use serde::{Deserialize, Serialize}; +use std::collections::HashSet; use std::path::{Path, PathBuf}; use crate::error::{KVError, Result}; @@ -32,6 +33,14 @@ pub struct Config { } // ===== Cluster / Placement ===== +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +pub struct RdmaRailAdvertiseConfig { + /// Stable fabric name shared with a client's local-path configuration. + pub fabric_id: String, + /// Reachable RDMA control listener, not a wildcard bind address. + pub listener_endpoint: String, +} + #[derive(Debug, Clone, Serialize, Deserialize, Default)] pub struct ClusterNodeConfig { /// Stable node ID. When empty the endpoint is used as fallback. @@ -42,6 +51,9 @@ pub struct ClusterNodeConfig { /// Optional RDMA endpoint, e.g. "10.0.0.11:18515". #[serde(default)] pub rdma_endpoint: String, + /// Optional alternative listeners on this storage node. + #[serde(default)] + pub rdma_rails: Vec, } #[derive(Debug, Clone, Serialize, Deserialize, Default)] @@ -55,6 +67,9 @@ pub struct ClusterConfig { /// This node's outward RDMA endpoint; if empty, read CS_RDMA_ADVERTISE. #[serde(default)] pub rdma_advertise: String, + /// Optional alternative listeners on this node. Does not affect object layout. + #[serde(default)] + pub rdma_rails: Vec, /// KVService data nodes eligible for object stripe placement. /// /// Empty means single-node mode; cross-node placement is enabled only when @@ -329,12 +344,14 @@ impl Config { "at least one storage device must be configured".to_string(), )); } + validate_rail_advertisements("cluster", &self.cluster.rdma_rails)?; for node in &self.cluster.data_nodes { if node.grpc_endpoint.trim().is_empty() { return Err(KVError::Config( "cluster.data_nodes.grpc_endpoint must not be empty".to_string(), )); } + validate_rail_advertisements("cluster.data_nodes", &node.rdma_rails)?; } match self.router.strategy.as_str() { "object_hash" => {} @@ -398,3 +415,60 @@ impl Config { Ok(()) } } + +fn validate_rail_advertisements(owner: &str, rails: &[RdmaRailAdvertiseConfig]) -> Result<()> { + let mut fabrics = HashSet::new(); + let mut listeners = HashSet::new(); + for rail in rails { + if rail.fabric_id.trim().is_empty() || rail.listener_endpoint.trim().is_empty() { + return Err(KVError::Config(format!( + "{owner}.rdma_rails require nonempty fabric_id and listener_endpoint" + ))); + } + if !fabrics.insert(rail.fabric_id.as_str()) + || !listeners.insert(rail.listener_endpoint.as_str()) + { + return Err(KVError::Config(format!( + "{owner}.rdma_rails require distinct fabrics and listeners" + ))); + } + } + Ok(()) +} + +#[cfg(test)] +mod rail_advertisement_tests { + use super::*; + + #[test] + fn duplicate_fabric_or_listener_is_rejected() { + let mut config = Config::default(); + config.cluster.rdma_rails = vec![ + RdmaRailAdvertiseConfig { + fabric_id: "fabric-a".into(), + listener_endpoint: "10.31.0.2:55153".into(), + }, + RdmaRailAdvertiseConfig { + fabric_id: "fabric-a".into(), + listener_endpoint: "10.32.0.2:55154".into(), + }, + ]; + assert!(config.validate().is_err()); + config.cluster.rdma_rails[1].fabric_id = "fabric-b".into(); + config.cluster.rdma_rails[1].listener_endpoint = "10.31.0.2:55153".into(); + assert!(config.validate().is_err()); + } + + #[test] + fn blank_fabric_or_listener_is_rejected() { + let mut config = Config::default(); + config.cluster.rdma_rails = vec![RdmaRailAdvertiseConfig { + fabric_id: " ".into(), + listener_endpoint: "10.31.0.2:55153".into(), + }]; + assert!(config.validate().is_err()); + config.cluster.rdma_rails[0].fabric_id = "fabric-a".into(); + config.cluster.rdma_rails[0].listener_endpoint = " ".into(); + assert!(config.validate().is_err()); + } +} From c1c1de161081a317c30837ee3e87a5fb8b3324d0 Mon Sep 17 00:00:00 2001 From: yanchaomei Date: Sat, 3 Oct 2026 04:10:39 +0800 Subject: [PATCH 20/20] Record two-RXE discovery and upgrade receipts Publish full command outputs, source-file hashes, and raw receipt SHA-256 values for discovered dual read, second-link failure/recovery, old-client/new-server and new-client/old-server compatibility, release builds, and two-node regression. --- .../results/softroce-vm-discovery.json | 121 ++++++++++++++++++ 1 file changed, 121 insertions(+) create mode 100644 kv-service/benchmarks/results/softroce-vm-discovery.json diff --git a/kv-service/benchmarks/results/softroce-vm-discovery.json b/kv-service/benchmarks/results/softroce-vm-discovery.json new file mode 100644 index 0000000..7736059 --- /dev/null +++ b/kv-service/benchmarks/results/softroce-vm-discovery.json @@ -0,0 +1,121 @@ +{ + "environment": "two isolated Ubuntu KVM guests; two independent virtio/tap/bridge/RXE paths; standard RDMA Verbs; one physical host and one virtual disk", + "interpretation": "functional, upgrade, and failure-boundary evidence only; these single reads are not a paired performance benchmark or physical HCA aggregation result", + "code_commit": "4e9217d822cba1ce795824ffa55188a26ed03efe", + "source_sha256": { + "kv-service/proto/kv_service.proto": "369b318bcd0eb67ee958cb67f6ff8e77b48c2327b976e96ae5f32b033c869ea7", + "kv-service/server/src/config.rs": "2f96753ff945d2073e9b1c6cd0f9edb70368ac587c4ff00a6a7fdda6c3a19a6b", + "kv-service/server/src/api/service.rs": "0d1125d650eca00a967f4c5c8bd80948a7c5195b338ad31cede5e205e7b42055", + "kv-service/client-rs/src/rail_read.rs": "83820515282f02609ac652b70b384c3300c6edcb56002cc24afb2c182421c10b", + "kv-service/client-rs/src/bin/rail_read_bench.rs": "160bc801bacd45318b238888638a6c4475675fc3b28d4044b3aff5b487b7415b", + "kv-service/client-rs/tests/rail_read_e2e.rs": "c99ac09031e8bc473bcb141c71df3fab6aabbe8628b948152b55fcf97391f745" + }, + "object": { + "namespace": "rust-bench", + "key": "autodisc202610030/__combined__", + "size_bytes": 67108864, + "stripes": 16, + "stripe_bytes": 4194304, + "xxh3": "a0a4cbfa5cad46af" + }, + "receipts": [ + { + "file": "seed-object.log", + "assertion": "A new checksummed 64 MiB object was written", + "raw_bytes": 412, + "raw_sha256": "06163ac559582426b6089a2f31419828e32a9f4e5e6a7158719c54a7326e46c6", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\n== Rust combined bench: 1 prefixes x 64 MB (single value, simulates __combined__ striping), concurrency=1, mode=stream(chunk=2MB) ==\n[PUT combined ] 1 x 64MB = 64MB 642.2ms 0.10 GB/s\n" + }, + { + "file": "object-metadata.log", + "assertion": "The stored object retained generation, layout version and 16/16 stripe checksums", + "raw_bytes": 3307, + "raw_sha256": "47bd9469a1d32aa0ddeb00743bdd23d1097597973c2f44babb5512dfa0714387", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nnamespace: rust-bench\nobject key: autodisc202610030/__combined__\ncanonical key: 10:rust-benchautodisc202610030/__combined__\nRedis key: contextstore:soft-roce-vm:block_meta:10:rust-benchautodisc202610030/__combined__\nsize: 64.00 MiB\ngeneration: 1\nlayout version: 1\ncontent etag: 5030219b4ea256c8\ncreated at: 2026-10-02T19:13:35Z\nlast accessed: 2026-10-02T19:13:35Z\nTTL seconds: 0\nstriping: 16 stripes, chunk size 4.00 MiB, checksum status complete (16/16)\nSTRIPE DEVICE CHECKSUM PATH\n 0 local:0 53c00d1575a12d12 /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk0.bin\n 1 local:1 7ae59d9bd34beb96 /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk1.bin\n 2 local:0 0225b8725399db9f /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk2.bin\n 3 local:1 4a7da3a96028cb70 /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk3.bin\n 4 local:0 c427e7a1784e2f55 /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk4.bin\n 5 local:1 c68f74e842437017 /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk5.bin\n 6 local:0 d61dd8edc3f14bc6 /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk6.bin\n 7 local:1 471f3d25188676ea /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk7.bin\n 8 local:0 181335f845768a56 /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk8.bin\n 9 local:1 e407641c4fdee312 /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk9.bin\n 10 local:0 a7ab4c41647be778 /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk10.bin\n 11 local:1 7ebd83707ac82d5f /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk11.bin\n 12 local:0 0247a18312bdd464 /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk12.bin\n 13 local:1 05d8c5f1d79fbe81 /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk13.bin\n 14 local:0 06354b7310fe639a /home/railtest/data/rail0/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk14.bin\n 15 local:1 c80fb0c3b3d0aa25 /home/railtest/data/rail1/contextstore/data/rust-bench/b2/9e/b082fbb926dad4e16ea08e03d94a.g1.l1.chunk15.bin\n" + }, + { + "file": "discovered-dual-read.log", + "assertion": "Two listener endpoints were discovered; both rails carried 32 MiB", + "raw_bytes": 1433, + "raw_sha256": "0a118524d0cc727045d1ef3826aaa4c8a992594c1bb8ad8b95b74e08cd65e29c", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r1,device=rxe_c1,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.32.0.2:55154\nsample,environment=soft-roce,rails=2,iteration=1,bytes=67108864,latency_us=580966,gib_per_s=0.108,xxh3=a0a4cbfa5cad46af\nsummary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=1,median_us=580966,cpu_user_us=64719,cpu_system_us=320566,peak_rss_kb=211268,rail_bytes=[33554432, 33554432]\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=356485,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=387277,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\n" + }, + { + "file": "discovered-e2e-success.log", + "assertion": "The real-Verbs discovered-rail E2E restored the complete object", + "raw_bytes": 387, + "raw_sha256": "0fd7bcf2b34d00a721296523452e1c47426c246c07abeb6a305b5333571d06e0", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\n\nrunning 1 test\ntest advertised_two_rails_restore_one_object ... ok\n\ntest result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 14 filtered out; finished in 1.48s\n\n" + }, + { + "file": "discovered-single-read.log", + "assertion": "One discovered rail retained the single-rail read path", + "raw_bytes": 965, + "raw_sha256": "c81e8168d28e8beaa4cfe94abf4b2b014c3aa11c78e48ca4e9ad59fe2502f0b0", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153\nsample,environment=soft-roce,rails=1,iteration=1,bytes=67108864,latency_us=419888,gib_per_s=0.149,xxh3=a0a4cbfa5cad46af\nsummary,environment=soft-roce,rails=1,bytes_per_iter=67108864,iters=1,median_us=419888,cpu_user_us=61037,cpu_system_us=79379,peak_rss_kb=202588,rail_bytes=[67108864]\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=67108864,avg_us=321615,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=67108864,registered_bytes_reserved=0,peak_registered_bytes=67108864\n" + }, + { + "file": "old-client-new-server.log", + "assertion": "An old protobuf client ignored the new optional field and read successfully", + "raw_bytes": 1276, + "raw_sha256": "7902b0d32634198f489ee1b30cbec4236b33f1acddd9ff276dd1f37e6fcc3d30", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nroute,id=r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153\nroute,id=r1,device=rxe_c1,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.32.0.2:55154\nsample,environment=soft-roce,rails=2,iteration=1,bytes=67108864,latency_us=393478,gib_per_s=0.159,xxh3=a0a4cbfa5cad46af\nsummary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=1,median_us=393478,cpu_user_us=66741,cpu_system_us=93785,peak_rss_kb=210664,rail_bytes=[33554432, 33554432]\nrail_stats,id=r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=294374,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\nrail_stats,id=r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=224631,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\n" + }, + { + "file": "discovery-link-fault-transcript.log", + "assertion": "Independent second-link DOWN, failure-atomic E2E, link UP and recovery", + "raw_bytes": 3334, + "raw_sha256": "ab5c516d631bfa8a98a4a41c37d9507fd4eddfbbe46a2c6745cca4c01d22ece2", + "full_output": "=== baseline ===\n** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nlink rxe_c0/1 state ACTIVE physical_state LINK_UP netdev enp0s3 \nlink rxe_c1/1 state ACTIVE physical_state LINK_UP netdev enp0s4 \n=== second local RXE link down ===\n** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nlink rxe_c0/1 state ACTIVE physical_state LINK_UP netdev enp0s3 \nlink rxe_c1/1 state DOWN physical_state DISABLED netdev enp0s4 \n=== fault-atomicity E2E ===\n** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\n\nrunning 1 test\ntest advertised_second_link_down_cannot_publish_partial_object ... ok\n\ntest result: ok. 1 passed; 0 failed; 0 ignored; 0 measured; 14 filtered out; finished in 4.53s\n\n=== restore second local RXE link ===\n** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nlink rxe_c0/1 state ACTIVE physical_state LINK_UP netdev enp0s3 \nlink rxe_c1/1 state ACTIVE physical_state LINK_UP netdev enp0s4 \nPING 10.32.0.2 (10.32.0.2) from 10.32.0.1 enp0s4: 56(84) bytes of data.\n64 bytes from 10.32.0.2: icmp_seq=1 ttl=64 time=1.09 ms\n\n--- 10.32.0.2 ping statistics ---\n1 packets transmitted, 1 received, 0% packet loss, time 0ms\nrtt min/avg/max/mdev = 1.088/1.088/1.088/0.000 ms\n=== normal discovered dual read after recovery ===\n** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r1,device=rxe_c1,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.32.0.2:55154\nsample,environment=soft-roce,rails=2,iteration=1,bytes=67108864,latency_us=403044,gib_per_s=0.155,xxh3=a0a4cbfa5cad46af\nsummary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=1,median_us=403044,cpu_user_us=80482,cpu_system_us=84570,peak_rss_kb=211128,rail_bytes=[33554432, 33554432]\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=297424,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=229543,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\n" + }, + { + "file": "new-client-old-server-manual.log", + "assertion": "The new client used explicit routes against the old server", + "raw_bytes": 1277, + "raw_sha256": "19e61840014b66cee7da40774b96b6098b86e680a722854279424a3a0801e4d4", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nroute,id=r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153\nroute,id=r1,device=rxe_c1,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.32.0.2:55154\nsample,environment=soft-roce,rails=2,iteration=1,bytes=67108864,latency_us=385329,gib_per_s=0.162,xxh3=a0a4cbfa5cad46af\nsummary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=1,median_us=385329,cpu_user_us=61534,cpu_system_us=158163,peak_rss_kb=210784,rail_bytes=[33554432, 33554432]\nrail_stats,id=r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=262168,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\nrail_stats,id=r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=265901,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\n" + }, + { + "file": "new-client-old-server-discovery-rejected.log", + "assertion": "Discovery against a server without capabilities failed explicitly", + "raw_bytes": 320, + "raw_sha256": "e9204650093635358b2a9329a0c0eb9868b83cee5330a0cce4a7d4828bad6a0e", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nError: invalid placement: placement has no rail capabilities; use manual routes for older servers\n" + }, + { + "file": "discovery-restored-after-upgrade.log", + "assertion": "Discovery worked again after restoring the new server", + "raw_bytes": 1432, + "raw_sha256": "cdc7f648b9039f6b05cb38e4efbc8cd2ce26dee296d51e4b8cfc35c9fb6cc531", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.31.0.2:55153\nroute,id=cs-soft-roce-vm-server/10.31.0.2:55153/r1,device=rxe_c1,port=1,gid=1,weight=1,advertised=10.31.0.2:55153,listener=10.32.0.2:55154\nsample,environment=soft-roce,rails=2,iteration=1,bytes=67108864,latency_us=361776,gib_per_s=0.173,xxh3=a0a4cbfa5cad46af\nsummary,environment=soft-roce,rails=2,bytes_per_iter=67108864,iters=1,median_us=361776,cpu_user_us=64857,cpu_system_us=82860,peak_rss_kb=211152,rail_bytes=[33554432, 33554432]\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r0,device=rxe_c0,listener=10.31.0.2:55153,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=269218,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\nrail_stats,id=cs-soft-roce-vm-server/10.31.0.2:55153/r1,device=rxe_c1,listener=10.32.0.2:55154,numa=None,pci=None,healthy=true,cooldown_ms=0,reads_ok=1,reads_err=0,bytes=33554432,avg_us=270546,inflight_requests=0,inflight_bytes=0,peak_inflight_bytes=33554432,registered_bytes_reserved=0,peak_registered_bytes=37748736\n" + }, + { + "file": "final-matrix-after-review.log", + "assertion": "Final RDMA server/client release suites, CLI tests, strict client Clippy and no-RDMA checks", + "raw_bytes": 1083, + "raw_sha256": "266902c14ed81e543e8569c1b9acf3613f5e42f40fc8576f301ddc0f3406c023", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\n\nrunning 102 tests\n....................................................................................... 87/102\n...............\ntest result: ok. 102 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.52s\n\n\nrunning 45 tests\n...........................i.................\ntest result: ok. 44 passed; 0 failed; 1 ignored; 0 measured; 0 filtered out; finished in 0.11s\n\n\nrunning 0 tests\n\ntest result: ok. 0 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s\n\n\nrunning 3 tests\n...\ntest result: ok. 3 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s\n\n\nrunning 2 tests\n..\ntest result: ok. 2 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 0.00s\n\n\nrunning 15 tests\niiiiiiiiiiiiiii\ntest result: ok. 0 passed; 0 failed; 15 ignored; 0 measured; 0 filtered out; finished in 0.00s\n\n" + }, + { + "file": "two-node-e2e.log", + "assertion": "Hardware-independent two-node Redis cluster regression after the protobuf change", + "raw_bytes": 6042, + "raw_sha256": "3694e105fe70cbc96ea169c6a1f089fffde4eada49286d185a4ed19b53b9a801", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\nmake -C kv-service e2e\nmake[1]: Entering directory '/home/kvgroup/chaomei/contextstore_multirail_20261002/kv-service'\n==> Building contextstore-server for hardware-independent E2E (release)\ncd server && cargo build --release --bin contextstore-server\n Finished `release` profile [optimized] target(s) in 0.16s\ncd client-rs && CS_E2E_SERVER_BIN=\"/home/kvgroup/chaomei/contextstore_multirail_20261002/kv-service/../target/release/contextstore-server\" \\\n\tcargo test --release --test local_cluster_e2e -- --nocapture --test-threads=1\n Finished `release` profile [optimized] target(s) in 0.14s\n Running tests/local_cluster_e2e.rs (/home/kvgroup/chaomei/contextstore_multirail_20261002/target/release/deps/local_cluster_e2e-b3619e7d914d69a1)\n\nrunning 4 tests\ntest two_node_cross_node_get_returns_original_payload ... e2e scenario=cross_node_get event=cluster_ready redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\ne2e scenario=cross_node_get event=node-a_client_connect redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\ne2e scenario=cross_node_get event=streaming_put_begin redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\ne2e scenario=cross_node_get event=streaming_put_complete redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\ne2e scenario=cross_node_get event=node-b_client_connect redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\ne2e scenario=cross_node_get event=cross_node_get_begin redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\ne2e scenario=cross_node_get event=cross_node_get_verified redis=127.0.0.1:36391 node_a=127.0.0.1:40963 node_b=127.0.0.1:35373 root=/tmp/.tmpBwR0lH\nok\ntest two_node_distributed_delete_retires_then_reclaims_all_stripe_files ... e2e scenario=distributed_delete event=cluster_ready redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=node-a_client_connect redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=streaming_put_begin redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=streaming_put_complete redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=placement_lookup_begin redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=placement_verified redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=distributed_delete_begin redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\ne2e scenario=distributed_delete event=distributed_delete_reclaimed redis=127.0.0.1:39419 node_a=127.0.0.1:45753 node_b=127.0.0.1:36877 root=/tmp/.tmppp6UcW\nok\ntest two_node_restart_recovers_shared_metadata_and_stripes ... e2e scenario=node_restart_recovery event=cluster_ready redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=node-a_client_connect redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=streaming_put_begin redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=streaming_put_complete redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=placement_lookup_begin redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=placement_verified redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=node_b_restart_begin redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=node_b_restart_complete redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=post_restart_get_begin redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\ne2e scenario=node_restart_recovery event=post_restart_get_verified redis=127.0.0.1:37249 node_a=127.0.0.1:34073 node_b=127.0.0.1:34351 root=/tmp/.tmptAQONU\nok\ntest two_node_streaming_put_places_stripes_on_all_devices ... e2e scenario=four_way_placement event=cluster_ready redis=127.0.0.1:40833 node_a=127.0.0.1:34825 node_b=127.0.0.1:43853 root=/tmp/.tmpfY3jqN\ne2e scenario=four_way_placement event=node-a_client_connect redis=127.0.0.1:40833 node_a=127.0.0.1:34825 node_b=127.0.0.1:43853 root=/tmp/.tmpfY3jqN\ne2e scenario=four_way_placement event=streaming_put_begin redis=127.0.0.1:40833 node_a=127.0.0.1:34825 node_b=127.0.0.1:43853 root=/tmp/.tmpfY3jqN\ne2e scenario=four_way_placement event=streaming_put_complete redis=127.0.0.1:40833 node_a=127.0.0.1:34825 node_b=127.0.0.1:43853 root=/tmp/.tmpfY3jqN\ne2e scenario=four_way_placement event=placement_lookup_begin redis=127.0.0.1:40833 node_a=127.0.0.1:34825 node_b=127.0.0.1:43853 root=/tmp/.tmpfY3jqN\ne2e scenario=four_way_placement event=placement_verified redis=127.0.0.1:40833 node_a=127.0.0.1:34825 node_b=127.0.0.1:43853 root=/tmp/.tmpfY3jqN\nok\n\ntest result: ok. 4 passed; 0 failed; 0 ignored; 0 measured; 0 filtered out; finished in 2.79s\n\nmake[1]: Leaving directory '/home/kvgroup/chaomei/contextstore_multirail_20261002/kv-service'\n" + }, + { + "file": "committed-code-build.log", + "assertion": "Release server/client binaries and real-RXE E2E executable built from committed code", + "raw_bytes": 612, + "raw_sha256": "ec7e2a9e770fcf0f98829b6c2b382e7d88844b4536d30da51d08624331095d84", + "full_output": "** WARNING: connection is not using a post-quantum key exchange algorithm.\r\n** This session may be vulnerable to \"store now, decrypt later\" attacks.\r\n** The server may need to be upgraded. See https://openssh.com/pq.html\r\n Compiling contextstore-server v0.1.0 (/home/kvgroup/chaomei/contextstore_multirail_20261002/kv-service/server)\n Finished `release` profile [optimized] target(s) in 1m 19s\n Finished `release` profile [optimized] target(s) in 0.15s\n Finished `release` profile [optimized] target(s) in 0.15s\n Executable tests/rail_read_e2e.rs (target/release/deps/rail_read_e2e-999328bb0f338845)\n" + }, + { + "file": "server-final.log", + "assertion": "Both isolated RXE listeners served real stripe-subset RDMA requests", + "raw_bytes": 4641, + "raw_sha256": "1edbded38ff46f9ca7d1bba58c368f13683f81eaf99e7d058c7477f0a16df146", + "full_output": "\u001b[2m2026-10-02T20:07:48.326014Z\u001b[0m \u001b[32m INFO\u001b[0m starting ContextStore KV Service v0.1.0...\n\u001b[2m2026-10-02T20:07:48.326109Z\u001b[0m \u001b[32m INFO\u001b[0m config file: /home/railtest/evidence/server.toml\n\u001b[2m2026-10-02T20:07:48.326279Z\u001b[0m \u001b[32m INFO\u001b[0m loaded config: 2 storage device(s), L1 capacity 128 MB, I/O executor = tier_a\n\u001b[2m2026-10-02T20:07:48.329630Z\u001b[0m \u001b[32m INFO\u001b[0m gRPC listening on: 10.31.0.2:55151\n\u001b[2m2026-10-02T20:07:48.329648Z\u001b[0m \u001b[32m INFO\u001b[0m CS_RDMA_SLAB_MB override rdma_slab_size_mb=128\n\u001b[2m2026-10-02T20:07:48.329655Z\u001b[0m \u001b[32m INFO\u001b[0m CS_RDMA_DEVICES override: 2 NIC(s) [\"rxe_s0@10.31.0.2:55153\", \"rxe_s1@10.32.0.2:55154\"]\n\u001b[2m2026-10-02T20:07:48.329659Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA tier enabled: 2 NIC(s), primary listener=10.31.0.2:55153\n\u001b[2m2026-10-02T20:07:48.331771Z\u001b[0m \u001b[32m INFO\u001b[0m RdmaContext: opened device=rxe_s0 port=1 gid_index=1 gid=[00, 00, 00, 00, 00, 00, 00, 00, 00, 00, ff, ff, 0a, 1f, 00, 02]\n\u001b[2m2026-10-02T20:07:48.331799Z\u001b[0m \u001b[32m INFO\u001b[0m opened NIC 0: dev=rxe_s0 port=1 gid_index=1\n\u001b[2m2026-10-02T20:07:48.332111Z\u001b[0m \u001b[32m INFO\u001b[0m RdmaContext: opened device=rxe_s1 port=1 gid_index=1 gid=[00, 00, 00, 00, 00, 00, 00, 00, 00, 00, ff, ff, 0a, 20, 00, 02]\n\u001b[2m2026-10-02T20:07:48.332125Z\u001b[0m \u001b[32m INFO\u001b[0m opened NIC 1: dev=rxe_s1 port=1 gid_index=1\n\u001b[2m2026-10-02T20:07:48.332165Z\u001b[0m \u001b[32m INFO\u001b[0m SlabBacking: mmap 134217728 bytes at 0x7f2074000000 (huge_pages=false)\n\u001b[2m2026-10-02T20:07:48.376175Z\u001b[0m \u001b[32m INFO\u001b[0m RdmaSlab registered: 134217728 bytes (128 MB) on 2 NIC(s), lkeys=[\"0x21a\", \"0x24e\"] rkeys=[\"0x21a\", \"0x24e\"]\n\u001b[2m2026-10-02T20:07:48.376218Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA slab injected into MemoryTier (128 MB, 2 NICs)\n\u001b[2m2026-10-02T20:07:48.376339Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA server listening on 10.31.0.2:55153 (nic_idx=0 dev=rxe_s0 port=1 gid_index=1)\n\u001b[2m2026-10-02T20:07:48.376568Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA server listening on 10.32.0.2:55154 (nic_idx=1 dev=rxe_s1 port=1 gid_index=1)\n\u001b[2m2026-10-02T20:07:51.463690Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA client connected: 10.31.0.1:54680 (nic_idx=0)\n\u001b[2m2026-10-02T20:07:51.463782Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA client connected: 10.32.0.1:56586 (nic_idx=1)\n\u001b[2m2026-10-02T20:07:51.464026Z\u001b[0m \u001b[32m INFO\u001b[0m RcQp created: qpn=17 psn=0x669043\n\u001b[2m2026-10-02T20:07:51.464113Z\u001b[0m \u001b[32m INFO\u001b[0m RcQp created: qpn=17 psn=0x67ead3\n\u001b[2m2026-10-02T20:07:51.464175Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA QP established: local_qpn=17 remote_qpn=17\n\u001b[2m2026-10-02T20:07:51.464292Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA QP established: local_qpn=17 remote_qpn=17\n\u001b[2m2026-10-02T20:07:51.724893Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA_SUBSET_DETAIL key=10:rust-benchautodisc202610030/__combined__ bytes=33554432 stripes=8 requests=32 alloc_us=4 stream_setup_us=10700 first_io_us=12453 last_io_us=16430 poll_us=219498 total_us=235921\n\u001b[2m2026-10-02T20:07:51.726369Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA_SUBSET_DETAIL key=10:rust-benchautodisc202610030/__combined__ bytes=33554432 stripes=8 requests=32 alloc_us=7 stream_setup_us=7868 first_io_us=8699 last_io_us=13231 poll_us=221432 total_us=234669\n\u001b[2m2026-10-02T20:08:21.328972Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA client connected: 10.31.0.1:36308 (nic_idx=0)\n\u001b[2m2026-10-02T20:08:21.329008Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA client connected: 10.32.0.1:55584 (nic_idx=1)\n\u001b[2m2026-10-02T20:08:21.329374Z\u001b[0m \u001b[32m INFO\u001b[0m RcQp created: qpn=17 psn=0x83a398\n\u001b[2m2026-10-02T20:08:21.329391Z\u001b[0m \u001b[32m INFO\u001b[0m RcQp created: qpn=17 psn=0x83e44f\n\u001b[2m2026-10-02T20:08:21.329637Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA QP established: local_qpn=17 remote_qpn=17\n\u001b[2m2026-10-02T20:08:21.329645Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA QP established: local_qpn=17 remote_qpn=17\n\u001b[2m2026-10-02T20:08:21.540169Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA_SUBSET_DETAIL key=10:rust-benchautodisc202610030/__combined__ bytes=33554432 stripes=8 requests=32 alloc_us=3 stream_setup_us=4558 first_io_us=6211 last_io_us=9699 poll_us=167599 total_us=177293\n\u001b[2m2026-10-02T20:08:21.615417Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA_SUBSET_DETAIL key=10:rust-benchautodisc202610030/__combined__ bytes=33554432 stripes=8 requests=32 alloc_us=5 stream_setup_us=1445 first_io_us=3012 last_io_us=12963 poll_us=246028 total_us=259002\n\u001b[2m2026-10-02T20:08:22.789962Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA client connected: 10.31.0.1:36322 (nic_idx=0)\n\u001b[2m2026-10-02T20:08:22.790359Z\u001b[0m \u001b[32m INFO\u001b[0m RcQp created: qpn=17 psn=0x987f29\n\u001b[2m2026-10-02T20:08:22.790643Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA QP established: local_qpn=17 remote_qpn=17\n\u001b[2m2026-10-02T20:08:23.104233Z\u001b[0m \u001b[32m INFO\u001b[0m RDMA_SUBSET_DETAIL key=10:rust-benchautodisc202610030/__combined__ bytes=67108864 stripes=16 requests=64 alloc_us=4 stream_setup_us=9616 first_io_us=11033 last_io_us=20239 poll_us=251926 total_us=272151\n" + } + ] +}