From d89040dae93137a616bd24509234c92c4f2dea55 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 05:07:29 +0000 Subject: [PATCH 01/18] feat(runtime): separate liveness, readiness, and degraded health Add a fake-clock health state machine with independent live/ready bits, recoverable stale_input/overload reasons, and sticky fatal/draining phases. Expose the snapshot on HealthHandle and an optional control_bind listener (/livez, /readyz, /health, /metrics) so supervisors can probe without a second server or tick-loop blocking. Co-authored-by: Raul Cardenas Montoya --- CHANGELOG.md | 5 + README.md | 10 +- docs/health.md | 109 ++++++ src/control.rs | 265 +++++++++++++ src/daemon.rs | 150 +++++++- src/health.rs | 983 +++++++++++++++++++++++++++++++++++++++++++++++++ src/lib.rs | 6 + 7 files changed, 1524 insertions(+), 4 deletions(-) create mode 100644 docs/health.md create mode 100644 src/control.rs create mode 100644 src/health.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 5c98976..c5d6add 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- Distinct liveness, readiness, recoverable degradation, and sticky fatal health + (`src/health.rs`) with a fake-clock state-machine test for every transition and + recovery path. Optional `control_bind` listener serves `/livez`, `/readyz`, + `/health`, and `/metrics` (the repository's first control surface; none existed + before). Contract, transition table, and example snapshots: [`docs/health.md`](docs/health.md). - GitHub Actions CI matrix: stub build/test/clippy on Linux, macOS, and Windows; rustfmt and optional `corpus-ipc` / libzmq jobs on Linux only (`docs/ci.md`). diff --git a/README.md b/README.md index b766678..c8de7c7 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,7 @@ Headless spiking neural-network runtime written in Rust. - Modular `neuromod::SpikingNetwork` core (CPU) - Optional **ZeroMQ PUB/SUB** networking via `corpus-ipc` - Headless **`brainstem-daemon`** binary for background execution +- Distinct **liveness / readiness / degraded / fatal** health snapshots (library handle plus optional `control_bind` listener) --- @@ -71,6 +72,10 @@ model_path = "~/models/soma16.mem" # literal path; `~` is not expanded tick_rate_hz = 1000 # loop frequency log_level = "info" # error|warn|info|debug|trace +# Optional process control surface (unset = no extra socket; historical default). +# Serves /livez, /readyz, /health, /metrics. See docs/health.md. +# control_bind = "127.0.0.1:9464" + # ZMQ (still required in TOML; no-ops under the default stub backend) spine_sub_port = 5555 # stimuli in spine_pub_port = 5556 # spikes out @@ -102,7 +107,9 @@ Default Cargo features are empty (`default = []` in `Cargo.toml`). That path use Enabling the feature does **not** change `BrainstemDaemon::new()` or `try_new()`. Those always inject `BackendPair::stub()`. Only `src/bin/brainstem_daemon.rs` constructs `ZmqStimulusSource` + `ZmqSpikeSink` when `corpus-ipc` is on. -Library users who want live ZMQ must build that pair themselves under `#[cfg(feature = "corpus-ipc")]` and pass it to `with_backend` / `try_with_backend`. Call `StimulusSource::initialize(...)` on the source first (as the binary does). Neither constructor nor `run` calls `initialize`; skipping it makes ingress fail with `ZmqBrainBackend not initialized`. +Library users who want live ZMQ must build that pair themselves under `#[cfg(feature = "corpus-ipc")]` and pass it to `with_backend` / `try_with_backend`. Call `StimulusSource::initialize(...)` on the source first (as the binary does). `run` also calls `initialize` (idempotent on success) so readiness can move past the checkpoint gate. Skipping initialize before `run` is therefore no longer required for the stub path; a failing `initialize` marks health **fatal** and never becomes ready. + +Health snapshots, probe paths, and the transition table live in [`docs/health.md`](docs/health.md). #### Config keys and env vars @@ -112,6 +119,7 @@ Library users who want live ZMQ must build that pair themselves under `#[cfg(fea | `tick_rate_hz` | used | used | | `log_level` | binary tracing init only; unused by `::new()` / `run` | binary tracing init only; unused by `::new()` / `run` | | `services` | used (`ServiceRegistry`) | used | +| `control_bind` | optional HTTP control surface; unset = no listener | same | | `spine_sub_port` | parsed, **no-op** | sets `SPIKENAUT_ZMQ_READOUT_IPC` to `tcp://127.0.0.1:` (also sets unused `CORPUS_IPC_ZMQ_READOUT_IPC` for compatibility) | | `spine_pub_port` | parsed, **no-op** | binds ZMQ PUB `tcp://*:` | | `model_path` | parsed, **no-op** (`StubStimulusSource::initialize` ignores it) | passed literally to `initialize` (no `~` expansion); pinned `ZmqBrainBackend` currently ignores `_model_path` | diff --git a/docs/health.md b/docs/health.md new file mode 100644 index 0000000..f0e40ed --- /dev/null +++ b/docs/health.md @@ -0,0 +1,109 @@ +# Runtime health + +Supervisors should treat **liveness** and **readiness** as independent. A live +`brainstem-daemon` process has a health reporter; it is ready only after +stimulus-source initialization and checkpoint validation have succeeded. + +This repository had no HTTP/metrics server before this surface. When +`control_bind` is set, `BrainstemDaemon::run` starts **one** listener: + +| Path | Meaning | +|---|---| +| `GET /livez` | `200` if live, `503` otherwise | +| `GET /readyz` | `200` if ready, `503` otherwise | +| `GET /health` | `200` JSON [`HealthSnapshot`](../src/health.rs) (always; inspect `phase`) | +| `GET /metrics` | Prometheus text; labels are phase/reason codes only | + +Leave `control_bind` unset to preserve the historical no-extra-socket default. +Do not add a second control server beside this one. + +Library embedders can also clone [`HealthHandle`](../src/health.rs) from +`BrainstemDaemon::health()` and call `snapshot()` / `try_snapshot()` without +waiting on the tick loop's backend or `SpikingNetwork::step`. + +## Transition table + +| From | Event | To | live | ready | Notes | +|---|---|---|---|---|---| +| (unstarted) | `ProcessStarted` | `starting` | true | false | Construction. Live does not imply ready. | +| `starting` | `InitializationCompleted` | `loading_checkpoint` | true | false | `initialize()` succeeded; checkpoint still required. | +| `starting` | `InitializationFailed` | `fatal` | true | false | Sticky. Detail is JSON-only, never a metric label. | +| `loading_checkpoint` | `CheckpointValidated` | `running` | true | true | Ready only after this gate. | +| `loading_checkpoint` | `CheckpointRejected` | `fatal` | true | false | Sticky. | +| `running` | clock ≥ `stale_after` without ingress | `degraded` | true | true | Reason `stale_input`. Ready stays true. | +| `degraded` (stale) | `IngressObserved` | `running` (if no other reasons) | true | true | Ticks without ingress do **not** clear stale. | +| `running` | queue fill ≥ `overload_high` | `degraded` | true | true | Reason `overload`. | +| `degraded` (overload) | fill ≤ `overload_low` | `running` (if no other reasons) | true | true | Hysteresis: mid-band does not recover. | +| `running` / `degraded` | `BeginDrain` | `draining` | true | false | SIGTERM/SIGINT. Does not return to ready. | +| any non-fatal | `Fatal` / init or checkpoint failure | `fatal` | true | false | Subsequent validate/tick/drain cannot restore ready. | + +Recoverable reasons (`stale_input`, `overload`) are independent: clearing one +leaves the other. `capacity == 0` means “no queue instrumented” (LIM-1216) and +never counts as overload. + +Fatal and draining are sticky for **this process**. A new process starts in +`starting` again. + +## Checkpoint stand-in + +[`LIM-1133`](https://linear.app/rpd-34/issue/LIM-1133) will load and digest a +real Spikenaut checkpoint. Until then, a successful `StimulusSource::initialize` +is treated as the checkpoint gate. The snapshot identity is the `model_path` +file name (not the full path) with `digest: null`. + +## Example snapshots + +Healthy (ready to consume events): + +```json +{ + "live": true, + "ready": true, + "phase": "running", + "reasons": [], + "last_successful_tick_ms": 0, + "checkpoint": { "id": "soma16", "digest": "abc123" }, + "input_freshness": { "age_ms": 0, "stale": false }, + "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, + "fatal": null, + "observed_at_ms": 0 +} +``` + +Degraded (still ready; supervisors should not bounce the process): + +```json +{ + "live": true, + "ready": true, + "phase": "degraded", + "reasons": ["stale_input", "overload"], + "last_successful_tick_ms": 0, + "checkpoint": { "id": "soma16", "digest": "abc123" }, + "input_freshness": { "age_ms": 100, "stale": true }, + "queue_pressure": { "depth": 95, "capacity": 100, "ratio": 0.95, "overloaded": true }, + "fatal": null, + "observed_at_ms": 100 +} +``` + +Fatal (never returns to ready in this process): + +```json +{ + "live": true, + "ready": false, + "phase": "fatal", + "reasons": ["fatal"], + "last_successful_tick_ms": null, + "checkpoint": null, + "input_freshness": { "age_ms": null, "stale": false }, + "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, + "fatal": { "code": "checkpoint_invalid", "detail": "blank weights" }, + "observed_at_ms": 0 +} +``` + +`fatal.detail` belongs in JSON/logs. Prometheus `/metrics` exposes +`brainstem_fatal 1` and `brainstem_phase{phase="fatal"} 1` without the detail +string. diff --git a/src/control.rs b/src/control.rs new file mode 100644 index 0000000..de095ba --- /dev/null +++ b/src/control.rs @@ -0,0 +1,265 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +//! Optional control surface for health snapshots and low-cardinality metrics. +//! +//! This repository previously had no HTTP/metrics server. A single listener is +//! started from `BrainstemDaemon::run` when `control_bind` is set. Do not add +//! a second server alongside this one. + +use std::net::SocketAddr; +use std::time::Duration; + +use anyhow::{Context, Result}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::{TcpListener, TcpStream}; +use tokio::sync::watch; +use tracing::{info, warn}; + +use crate::health::{HealthHandle, HealthSnapshot}; + +pub async fn serve( + addr: SocketAddr, + health: HealthHandle, + shutdown: watch::Receiver, +) -> Result<()> { + let listener = TcpListener::bind(addr) + .await + .with_context(|| format!("failed to bind control surface on {addr}"))?; + serve_listener(listener, health, shutdown).await +} + +pub async fn serve_listener( + listener: TcpListener, + health: HealthHandle, + mut shutdown: watch::Receiver, +) -> Result<()> { + let bound = listener + .local_addr() + .context("control listener has no local address")?; + info!(%bound, "control surface listening (/livez /readyz /health /metrics)"); + + loop { + tokio::select! { + changed = shutdown.changed() => { + if changed.is_err() || *shutdown.borrow() { + break; + } + } + accepted = listener.accept() => { + match accepted { + Ok((stream, _)) => { + let health = health.clone(); + tokio::spawn(async move { + if let Err(e) = handle_connection(stream, &health).await { + warn!("control connection failed: {e}"); + } + }); + } + Err(e) => warn!("control accept failed: {e}"), + } + } + } + } + + Ok(()) +} + +async fn handle_connection(mut stream: TcpStream, health: &HealthHandle) -> Result<()> { + let mut buf = [0u8; 1024]; + let n = tokio::time::timeout(Duration::from_secs(2), stream.read(&mut buf)) + .await + .context("control read timed out")? + .context("control read failed")?; + let req = std::str::from_utf8(&buf[..n]).unwrap_or(""); + let snap = health.snapshot(); + let response = render_http(req, &snap); + stream.write_all(&response).await?; + stream.flush().await?; + Ok(()) +} + +pub(crate) fn render_http(request: &str, snap: &HealthSnapshot) -> Vec { + match parse_get_path(request) { + ParseResult::Get(path) => { + let (status, content_type, body) = match path { + "/livez" | "/healthz/live" => probe(snap.live, b"live\n", b"not live\n"), + "/readyz" | "/healthz/ready" => probe(snap.ready, b"ready\n", b"not ready\n"), + "/health" => ( + 200, + "application/json", + serde_json::to_vec(snap).unwrap_or_else(|_| b"{}".to_vec()), + ), + "/metrics" => ( + 200, + "text/plain; version=0.0.4", + snap.prometheus_text().into_bytes(), + ), + _ => (404, "text/plain; charset=utf-8", b"not found\n".to_vec()), + }; + http_response(status, content_type, &body) + } + ParseResult::NotGet => { + http_response(405, "text/plain; charset=utf-8", b"method not allowed\n") + } + ParseResult::Invalid => http_response(400, "text/plain; charset=utf-8", b"bad request\n"), + } +} + +fn probe(ok: bool, yes: &'static [u8], no: &'static [u8]) -> (u16, &'static str, Vec) { + if ok { + (200, "text/plain; charset=utf-8", yes.to_vec()) + } else { + (503, "text/plain; charset=utf-8", no.to_vec()) + } +} + +enum ParseResult<'a> { + Get(&'a str), + NotGet, + Invalid, +} + +fn parse_get_path(request: &str) -> ParseResult<'_> { + let line = match request.lines().next() { + Some(line) => line, + None => return ParseResult::Invalid, + }; + let mut parts = line.split_whitespace(); + let method = match parts.next() { + Some(method) => method, + None => return ParseResult::Invalid, + }; + let target = match parts.next() { + Some(target) => target, + None => return ParseResult::Invalid, + }; + if !method.eq_ignore_ascii_case("GET") { + return ParseResult::NotGet; + } + let path = target.split('?').next().unwrap_or(target); + ParseResult::Get(path) +} + +fn http_response(status: u16, content_type: &str, body: &[u8]) -> Vec { + let reason = match status { + 200 => "OK", + 400 => "Bad Request", + 404 => "Not Found", + 405 => "Method Not Allowed", + 503 => "Service Unavailable", + _ => "OK", + }; + let mut out = Vec::new(); + out.extend_from_slice( + format!( + "HTTP/1.1 {status} {reason}\r\nContent-Type: {content_type}\r\nContent-Length: {}\r\nConnection: close\r\n\r\n", + body.len() + ) + .as_bytes(), + ); + out.extend_from_slice(body); + out +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::health::{ + CheckpointIdentity, FakeClock, HealthEvent, HealthHandle, HealthLimits, HealthMachine, + ReasonCode, + }; + + fn ready_snap() -> HealthSnapshot { + let clock = FakeClock::new(); + let mut machine = HealthMachine::new(clock, HealthLimits::default()); + machine.apply(HealthEvent::ProcessStarted); + machine.apply(HealthEvent::InitializationCompleted); + machine.apply(HealthEvent::CheckpointValidated { + identity: CheckpointIdentity { + id: "soma16".into(), + digest: None, + }, + }); + machine.apply(HealthEvent::IngressObserved); + machine.snapshot() + } + + #[test] + fn livez_and_readyz_use_status_codes() { + let snap = ready_snap(); + let live = String::from_utf8(render_http("GET /livez HTTP/1.1\r\n\r\n", &snap)).unwrap(); + assert!(live.starts_with("HTTP/1.1 200 OK")); + let ready = String::from_utf8(render_http("GET /readyz HTTP/1.1\r\n\r\n", &snap)).unwrap(); + assert!(ready.starts_with("HTTP/1.1 200 OK")); + + let starting = HealthMachine::new(FakeClock::new(), HealthLimits::default()); + let mut starting_m = starting; + starting_m.apply(HealthEvent::ProcessStarted); + let starting = starting_m.snapshot(); + let live = + String::from_utf8(render_http("GET /livez HTTP/1.1\r\n\r\n", &starting)).unwrap(); + assert!(live.starts_with("HTTP/1.1 200 OK")); + let ready = + String::from_utf8(render_http("GET /readyz HTTP/1.1\r\n\r\n", &starting)).unwrap(); + assert!(ready.starts_with("HTTP/1.1 503")); + assert!(!starting.ready); + assert!(starting.reasons.contains(&ReasonCode::Starting)); + } + + #[test] + fn health_json_is_always_200() { + let mut machine = HealthMachine::new(FakeClock::new(), HealthLimits::default()); + machine.apply(HealthEvent::ProcessStarted); + let snap = machine.snapshot(); + let raw = String::from_utf8(render_http("GET /health HTTP/1.1\r\n\r\n", &snap)).unwrap(); + assert!(raw.starts_with("HTTP/1.1 200 OK")); + assert!(raw.contains("\"live\":true")); + assert!(raw.contains("\"ready\":false")); + } + + #[test] + fn unknown_path_and_method() { + let snap = ready_snap(); + let not_found = + String::from_utf8(render_http("GET /nope HTTP/1.1\r\n\r\n", &snap)).unwrap(); + assert!(not_found.starts_with("HTTP/1.1 404")); + let bad_method = + String::from_utf8(render_http("POST /health HTTP/1.1\r\n\r\n", &snap)).unwrap(); + assert!(bad_method.starts_with("HTTP/1.1 405")); + } + + #[tokio::test] + async fn livez_is_200_while_readyz_is_503_before_checkpoint() { + let health = HealthHandle::started(HealthLimits::default()); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let addr = listener.local_addr().unwrap(); + let (tx, rx) = watch::channel(false); + let server = tokio::spawn(async move { serve_listener(listener, health, rx).await }); + + let live = http_get(addr, "/livez").await; + assert!(live.contains("HTTP/1.1 200"), "{live}"); + let ready = http_get(addr, "/readyz").await; + assert!(ready.contains("HTTP/1.1 503"), "{ready}"); + let body = http_get(addr, "/health").await; + assert!(body.contains("\"live\":true")); + assert!(body.contains("\"ready\":false")); + + let _ = tx.send(true); + server.await.unwrap().unwrap(); + } + + async fn http_get(addr: SocketAddr, path: &str) -> String { + let mut stream = tokio::net::TcpStream::connect(addr).await.unwrap(); + stream + .write_all(format!("GET {path} HTTP/1.1\r\nHost: localhost\r\n\r\n").as_bytes()) + .await + .unwrap(); + let mut buf = vec![0u8; 2048]; + let n = tokio::time::timeout(Duration::from_secs(2), stream.read(&mut buf)) + .await + .unwrap() + .unwrap(); + String::from_utf8_lossy(&buf[..n]).into_owned() + } +} diff --git a/src/daemon.rs b/src/daemon.rs index bd8ccb5..efec49c 100644 --- a/src/daemon.rs +++ b/src/daemon.rs @@ -4,19 +4,22 @@ //! Brainstem daemon runtime and config-driven service registry. use std::fs; -use std::path::PathBuf; +use std::net::SocketAddr; +use std::path::{Path, PathBuf}; use std::time::{Duration, SystemTime, UNIX_EPOCH}; use anyhow::{Context, Result, bail}; use neuromod::{NeuroModulators, SpikingNetwork}; use serde::Deserialize; use tokio::signal; +use tokio::sync::watch; use tokio::time; use tracing::{error, info, warn}; use crate::backend::{ BackendPair, IngressPacket, SpikeEvent as LocalSpikeEvent, SpikeSink, StimulusSource, }; +use crate::health::{CheckpointIdentity, HealthEvent, HealthHandle, HealthLimits, HealthSnapshot}; use crate::registry::{ServiceConfig, ServiceRegistry}; // Keep the const for compatibility when the corpus-ipc feature is used. @@ -35,6 +38,12 @@ pub struct DaemonConfig { pub channels: usize, #[serde(default)] pub services: Vec, + /// Optional `ip:port` for the process control surface (`/livez`, `/readyz`, `/health`, `/metrics`). + /// + /// Unset by default so existing configs keep opening no extra sockets. This is the + /// repository's only HTTP listener; do not add a second server beside it. + #[serde(default)] + pub control_bind: Option, } impl DaemonConfig { @@ -68,6 +77,7 @@ pub struct BrainstemDaemon { config: DaemonConfig, registry: ServiceRegistry, backend: BackendPair, + health: HealthHandle, } impl BrainstemDaemon { @@ -118,6 +128,7 @@ impl BrainstemDaemon { config, registry, backend, + health: HealthHandle::started(HealthLimits::default()), }) } @@ -126,15 +137,50 @@ impl BrainstemDaemon { &self.registry } + /// Clone the non-blocking health handle (independent of the tick-loop backend lock). + pub fn health(&self) -> HealthHandle { + self.health.clone() + } + + /// Current health snapshot. Does not wait on ingress or `SpikingNetwork::step`. + pub fn health_snapshot(&self) -> HealthSnapshot { + self.health.snapshot() + } + /// Run the daemon until a termination signal is received. pub async fn run(self) -> Result<()> { let cfg = self.config; let mut backend = self.backend; + let health = self.health; if cfg.tick_rate_hz == 0 || cfg.tick_rate_hz > 1_000_000 { anyhow::bail!("tick_rate_hz must be in range 1..=1_000_000"); } + let (control_stop, control_task) = + start_control(cfg.control_bind.as_deref(), health.clone()).await?; + + let model_path = cfg.model_path.to_string_lossy(); + if let Err(e) = backend.source.initialize(Some(model_path.as_ref())) { + health.apply(HealthEvent::InitializationFailed { + detail: e.to_string(), + }); + error!("Stimulus source initialization failed: {e}"); + await_shutdown_if_control(control_stop.is_some()).await; + stop_control(control_stop, control_task).await; + return Err(e).context("failed to initialize stimulus source"); + } + health.apply(HealthEvent::InitializationCompleted); + + // Successful initialize is the current checkpoint gate. Real digest/weight + // validation is tracked in LIM-1133 and will replace this stand-in. + health.apply(HealthEvent::CheckpointValidated { + identity: CheckpointIdentity { + id: checkpoint_id_from_path(&cfg.model_path), + digest: None, + }, + }); + let tick_duration = Duration::from_nanos(1_000_000_000 / u64::from(cfg.tick_rate_hz)); let mut ticker = time::interval(tick_duration); ticker.set_missed_tick_behavior(time::MissedTickBehavior::Skip); @@ -155,15 +201,19 @@ impl BrainstemDaemon { &mut *backend.sink, &mut stimuli, &mut spike_buf, + &health, ); } _ = &mut shutdown => { info!("Termination signal received, shutting down"); + health.apply(HealthEvent::BeginDrain); break; } } } + stop_control(control_stop, control_task).await; + // Explicit backend lifecycle hooks (flush sink, shutdown source) are invoked // for custom backends. Current built-ins are no-ops, but this satisfies // CodeAnt/CodeRabbit "missing cleanup" notes. @@ -223,6 +273,54 @@ fn init_runtime_default() -> BackendPair { BackendPair::stub() } +async fn start_control( + bind: Option<&str>, + health: HealthHandle, +) -> Result<( + Option>, + Option>, +)> { + let Some(bind) = bind else { + return Ok((None, None)); + }; + let addr: SocketAddr = bind + .parse() + .with_context(|| format!("invalid control_bind {bind}"))?; + let (tx, rx) = watch::channel(false); + let task = tokio::spawn(async move { + if let Err(e) = crate::control::serve(addr, health, rx).await { + warn!("control surface stopped: {e}"); + } + }); + Ok((Some(tx), Some(task))) +} + +async fn stop_control( + stop: Option>, + task: Option>, +) { + if let Some(tx) = stop { + let _ = tx.send(true); + } + if let Some(task) = task { + let _ = task.await; + } +} + +async fn await_shutdown_if_control(has_control: bool) { + if has_control { + shutdown_signal().await; + } +} + +fn checkpoint_id_from_path(path: &Path) -> String { + path.file_name() + .and_then(|name| name.to_str()) + .filter(|name| !name.is_empty()) + .unwrap_or("unknown") + .to_string() +} + fn validate_neuron_count(config: &DaemonConfig) -> Result<()> { let total = config .lif_count @@ -255,9 +353,13 @@ fn run_tick( sink: &mut dyn SpikeSink, stimuli: &mut [f32], spike_buf: &mut Vec, + health: &HealthHandle, ) { let packet = match source.next_ingress() { - Ok(Some(p)) => p, + Ok(Some(p)) => { + health.apply(HealthEvent::IngressObserved); + p + } Ok(None) => { // Per StimulusSource contract: None means skip ingress this tick but still // advance the network with zeroed stimuli (maintains tick cadence). @@ -284,6 +386,7 @@ fn run_tick( return; } }; + health.apply(HealthEvent::TickSucceeded); // Single timestamp for both per-spike time and batch metadata (keeps them consistent). let now = SystemTime::now() @@ -368,8 +471,9 @@ pub(crate) fn run_tick_for_test( sink: &mut dyn SpikeSink, stimuli: &mut [f32], spike_buf: &mut Vec, + health: &HealthHandle, ) { - run_tick(source, network, sink, stimuli, spike_buf); + run_tick(source, network, sink, stimuli, spike_buf, health); } #[cfg(test)] @@ -391,6 +495,7 @@ mod tests { ServiceConfig::named("telemetry"), ServiceConfig::named("critic-ipc"), ], + control_bind: None, } } @@ -402,6 +507,36 @@ mod tests { assert!(daemon.registry().contains("critic-ipc")); } + #[test] + #[test] + fn daemon_is_live_not_ready_before_run() { + let daemon = BrainstemDaemon::new(sample_config()); + let snap = daemon.health_snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, crate::health::HealthPhase::Starting); + assert!(!snap.reasons.is_empty()); + } + + #[test] + fn config_parses_optional_control_bind() { + let cfg: DaemonConfig = toml::from_str( + r#" +tick_rate_hz = 1000 +log_level = "info" +spine_sub_port = 5555 +spine_pub_port = 5556 +model_path = "/tmp/model.mem" +lif_count = 1 +izh_count = 0 +channels = 1 +control_bind = "127.0.0.1:9464" +"#, + ) + .expect("toml"); + assert_eq!(cfg.control_bind.as_deref(), Some("127.0.0.1:9464")); + } + #[test] fn daemon_ignores_disabled_services() { let mut cfg = sample_config(); @@ -506,6 +641,7 @@ mod tests { let mut network = SpikingNetwork::with_dimensions(2, 0, 2); let mut stimuli = vec![0.0; 2]; let mut spike_buf: Vec = Vec::new(); + let health = HealthHandle::started(HealthLimits::default()); // Prime one tick run_tick_for_test( @@ -514,10 +650,18 @@ mod tests { &mut sink, &mut stimuli, &mut spike_buf, + &health, ); // Sink should have received one (possibly empty) batch assert_eq!(sink.emitted.len(), 1); + let snap = health.snapshot(); + assert!(snap.live); + assert!( + snap.last_successful_tick_ms.is_some(), + "a successful network step must update last_successful_tick" + ); + assert!(!snap.input_freshness.stale); } // Sends a real SIGTERM to this test process, so it's `#[ignore]`d by default: diff --git a/src/health.rs b/src/health.rs new file mode 100644 index 0000000..0bdaf00 --- /dev/null +++ b/src/health.rs @@ -0,0 +1,983 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +//! Process health: liveness, readiness, recoverable degradation, and sticky fatal state. +//! +//! Supervisors should treat [`HealthSnapshot::live`] and [`HealthSnapshot::ready`] as +//! independent. A live process has not necessarily loaded a valid checkpoint. Recoverable +//! reasons (`stale_input`, `overload`) clear only after the condition is observed healthy. +//! Fatal and draining states never return to ready in the same process. +//! +//! Snapshot reads take a separate lock from the tick loop's backend and network, so they +//! do not wait on `StimulusSource` or `SpikingNetwork::step`. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, RwLock}; +use std::time::{Duration, Instant}; + +use serde::{Deserialize, Serialize}; + +/// Monotonic clock used by the health state machine. +pub trait Clock: Send + Sync { + fn now(&self) -> Instant; +} + +/// Wall-clock monotonic clock. +#[derive(Debug, Clone, Copy, Default)] +pub struct SystemClock; + +impl Clock for SystemClock { + fn now(&self) -> Instant { + Instant::now() + } +} + +/// Test clock. Clone and share the same offset with a [`HealthMachine`]. +#[derive(Debug, Clone)] +pub struct FakeClock { + origin: Instant, + offset_nanos: Arc, +} + +impl FakeClock { + pub fn new() -> Self { + Self { + origin: Instant::now(), + offset_nanos: Arc::new(AtomicU64::new(0)), + } + } + + pub fn advance(&self, duration: Duration) { + let add = u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX); + self.offset_nanos.fetch_add(add, Ordering::SeqCst); + } +} + +impl Default for FakeClock { + fn default() -> Self { + Self::new() + } +} + +impl Clock for FakeClock { + fn now(&self) -> Instant { + self.origin + Duration::from_nanos(self.offset_nanos.load(Ordering::SeqCst)) + } +} + +/// Thresholds for recoverable degradation. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct HealthLimits { + /// Ingress older than this marks `stale_input`. + pub stale_after: Duration, + /// Queue fill ratio that *enters* overload (`depth / capacity`). + pub overload_high: f64, + /// Queue fill ratio that *exits* overload (hysteresis; must be `<= overload_high`). + pub overload_low: f64, +} + +impl Default for HealthLimits { + fn default() -> Self { + Self { + stale_after: Duration::from_millis(500), + overload_high: 0.90, + overload_low: 0.70, + } + } +} + +/// Coarse phase derived from the snapshot. Stable, low-cardinality. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum HealthPhase { + Starting, + LoadingCheckpoint, + Running, + Degraded, + Draining, + Fatal, +} + +impl HealthPhase { + pub const ALL: [HealthPhase; 6] = [ + HealthPhase::Starting, + HealthPhase::LoadingCheckpoint, + HealthPhase::Running, + HealthPhase::Degraded, + HealthPhase::Draining, + HealthPhase::Fatal, + ]; + + pub fn as_str(self) -> &'static str { + match self { + HealthPhase::Starting => "starting", + HealthPhase::LoadingCheckpoint => "loading_checkpoint", + HealthPhase::Running => "running", + HealthPhase::Degraded => "degraded", + HealthPhase::Draining => "draining", + HealthPhase::Fatal => "fatal", + } + } +} + +/// Stable reason codes. Never put detailed error text in metric labels — use these codes. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ReasonCode { + Starting, + CheckpointPending, + StaleInput, + Overload, + Draining, + Fatal, +} + +impl ReasonCode { + pub const RECOVERABLE: [ReasonCode; 2] = [ReasonCode::StaleInput, ReasonCode::Overload]; + + pub fn as_str(self) -> &'static str { + match self { + ReasonCode::Starting => "starting", + ReasonCode::CheckpointPending => "checkpoint_pending", + ReasonCode::StaleInput => "stale_input", + ReasonCode::Overload => "overload", + ReasonCode::Draining => "draining", + ReasonCode::Fatal => "fatal", + } + } +} + +/// Sticky fatal class. Low-cardinality; details live on [`FatalState::detail`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum FatalCode { + InitializationFailed, + CheckpointInvalid, + Unspecified, +} + +impl FatalCode { + pub fn as_str(self) -> &'static str { + match self { + FatalCode::InitializationFailed => "initialization_failed", + FatalCode::CheckpointInvalid => "checkpoint_invalid", + FatalCode::Unspecified => "unspecified", + } + } +} + +/// Identity of the loaded checkpoint. Digest may be absent until real validation lands. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct CheckpointIdentity { + pub id: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub digest: Option, +} + +/// Fatal snapshot payload. `detail` is for logs/JSON, never a Prometheus label. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct FatalState { + pub code: FatalCode, + pub detail: String, +} + +/// Ingress freshness relative to the health clock. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct InputFreshness { + pub age_ms: Option, + pub stale: bool, +} + +/// Bounded-queue pressure. `capacity == 0` means "no queue instrumented". +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct QueuePressure { + pub depth: u64, + pub capacity: u64, + pub ratio: Option, + pub overloaded: bool, +} + +/// Machine-readable health view for supervisors and the control surface. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct HealthSnapshot { + pub live: bool, + pub ready: bool, + pub phase: HealthPhase, + pub reasons: Vec, + pub last_successful_tick_ms: Option, + pub checkpoint: Option, + pub input_freshness: InputFreshness, + pub queue_pressure: QueuePressure, + pub fatal: Option, + pub observed_at_ms: u64, +} + +impl HealthSnapshot { + /// Prometheus text exposition. Labels are stable reason/phase codes only. + pub fn prometheus_text(&self) -> String { + let mut out = String::new(); + out.push_str("# HELP brainstem_live 1 if the process health reporter is running.\n"); + out.push_str("# TYPE brainstem_live gauge\n"); + out.push_str(&format!("brainstem_live {}\n", u8::from(self.live))); + + out.push_str( + "# HELP brainstem_ready 1 if initialized, checkpoint-valid, not draining, not fatal.\n", + ); + out.push_str("# TYPE brainstem_ready gauge\n"); + out.push_str(&format!("brainstem_ready {}\n", u8::from(self.ready))); + + out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); + out.push_str("# TYPE brainstem_phase gauge\n"); + for phase in HealthPhase::ALL { + out.push_str(&format!( + "brainstem_phase{{phase=\"{}\"}} {}\n", + phase.as_str(), + u8::from(self.phase == phase) + )); + } + + out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); + out.push_str("# TYPE brainstem_degraded gauge\n"); + for reason in ReasonCode::RECOVERABLE { + let active = self.reasons.contains(&reason); + out.push_str(&format!( + "brainstem_degraded{{reason=\"{}\"}} {}\n", + reason.as_str(), + u8::from(active) + )); + } + + out.push_str( + "# HELP brainstem_fatal 1 if this process has entered a sticky fatal state.\n", + ); + out.push_str("# TYPE brainstem_fatal gauge\n"); + out.push_str(&format!( + "brainstem_fatal {}\n", + u8::from(self.fatal.is_some()) + )); + + out.push_str( + "# HELP brainstem_last_successful_tick_ms Milliseconds since start of last successful tick.\n", + ); + out.push_str("# TYPE brainstem_last_successful_tick_ms gauge\n"); + out.push_str(&format!( + "brainstem_last_successful_tick_ms {}\n", + self.last_successful_tick_ms.unwrap_or(0) + )); + + out.push_str("# HELP brainstem_input_age_ms Age of last ingress sample in milliseconds.\n"); + out.push_str("# TYPE brainstem_input_age_ms gauge\n"); + out.push_str(&format!( + "brainstem_input_age_ms {}\n", + self.input_freshness.age_ms.unwrap_or(0) + )); + + out.push_str("# HELP brainstem_queue_depth Ingress queue depth.\n"); + out.push_str("# TYPE brainstem_queue_depth gauge\n"); + out.push_str(&format!( + "brainstem_queue_depth {}\n", + self.queue_pressure.depth + )); + + out.push_str("# HELP brainstem_queue_capacity Ingress queue capacity.\n"); + out.push_str("# TYPE brainstem_queue_capacity gauge\n"); + out.push_str(&format!( + "brainstem_queue_capacity {}\n", + self.queue_pressure.capacity + )); + + out + } +} + +/// State-machine events. Tick-loop I/O never runs while these are applied. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HealthEvent { + ProcessStarted, + InitializationCompleted, + InitializationFailed { detail: String }, + CheckpointValidated { identity: CheckpointIdentity }, + CheckpointRejected { detail: String }, + TickSucceeded, + IngressObserved, + QueuePressure { depth: u64, capacity: u64 }, + BeginDrain, + Fatal { code: FatalCode, detail: String }, +} + +/// Pure health state machine. Drive it with [`FakeClock`] in tests. +pub struct HealthMachine { + clock: Arc, + limits: HealthLimits, + started: bool, + initialized: bool, + checkpoint_ok: bool, + draining: bool, + started_at: Option, + checkpoint_at: Option, + last_tick: Option, + last_ingress: Option, + checkpoint: Option, + queue_depth: u64, + queue_capacity: u64, + overloaded: bool, + fatal: Option, +} + +impl HealthMachine { + pub fn new(clock: impl Clock + 'static, limits: HealthLimits) -> Self { + Self { + clock: Arc::new(clock), + limits, + started: false, + initialized: false, + checkpoint_ok: false, + draining: false, + started_at: None, + checkpoint_at: None, + last_tick: None, + last_ingress: None, + checkpoint: None, + queue_depth: 0, + queue_capacity: 0, + overloaded: false, + fatal: None, + } + } + + pub fn apply(&mut self, event: HealthEvent) { + let now = self.clock.now(); + match event { + HealthEvent::ProcessStarted => { + if !self.started { + self.started = true; + self.started_at = Some(now); + } + } + HealthEvent::InitializationCompleted => { + if self.fatal.is_some() || self.draining { + return; + } + self.initialized = true; + } + HealthEvent::InitializationFailed { detail } => { + self.enter_fatal(FatalCode::InitializationFailed, detail); + } + HealthEvent::CheckpointValidated { identity } => { + if self.fatal.is_some() || self.draining { + return; + } + if !self.initialized { + return; + } + self.checkpoint = Some(identity); + self.checkpoint_ok = true; + self.checkpoint_at = Some(now); + } + HealthEvent::CheckpointRejected { detail } => { + self.enter_fatal(FatalCode::CheckpointInvalid, detail); + } + HealthEvent::TickSucceeded => { + self.last_tick = Some(now); + } + HealthEvent::IngressObserved => { + self.last_ingress = Some(now); + } + HealthEvent::QueuePressure { depth, capacity } => { + self.queue_depth = depth; + self.queue_capacity = capacity; + self.overloaded = next_overload( + self.overloaded, + depth, + capacity, + self.limits.overload_high, + self.limits.overload_low, + ); + } + HealthEvent::BeginDrain => { + if self.fatal.is_some() { + return; + } + self.draining = true; + } + HealthEvent::Fatal { code, detail } => { + self.enter_fatal(code, detail); + } + } + } + + pub fn snapshot(&self) -> HealthSnapshot { + let now = self.clock.now(); + let origin = self.started_at.unwrap_or(now); + let ms = |t: Instant| duration_ms(t.saturating_duration_since(origin)); + let observed_at_ms = ms(now); + + let age_ms = self + .last_ingress + .map(|t| duration_ms(now.saturating_duration_since(t))); + let stale = self.compute_stale(now); + let live = self.started; + let ready = live + && self.initialized + && self.checkpoint_ok + && !self.draining + && self.fatal.is_none(); + + let mut reasons = Vec::new(); + if self.fatal.is_some() { + reasons.push(ReasonCode::Fatal); + } else { + if !self.initialized { + reasons.push(ReasonCode::Starting); + } else if !self.checkpoint_ok { + reasons.push(ReasonCode::CheckpointPending); + } + if self.draining { + reasons.push(ReasonCode::Draining); + } + if stale { + reasons.push(ReasonCode::StaleInput); + } + if self.overloaded { + reasons.push(ReasonCode::Overload); + } + } + + let phase = if self.fatal.is_some() { + HealthPhase::Fatal + } else if self.draining { + HealthPhase::Draining + } else if !self.initialized { + HealthPhase::Starting + } else if !self.checkpoint_ok { + HealthPhase::LoadingCheckpoint + } else if stale || self.overloaded { + HealthPhase::Degraded + } else { + HealthPhase::Running + }; + + let ratio = if self.queue_capacity == 0 { + None + } else { + Some(self.queue_depth as f64 / self.queue_capacity as f64) + }; + + HealthSnapshot { + live, + ready, + phase, + reasons, + last_successful_tick_ms: self.last_tick.map(ms), + checkpoint: self.checkpoint.clone(), + input_freshness: InputFreshness { age_ms, stale }, + queue_pressure: QueuePressure { + depth: self.queue_depth, + capacity: self.queue_capacity, + ratio, + overloaded: self.overloaded, + }, + fatal: self.fatal.clone(), + observed_at_ms, + } + } + + fn enter_fatal(&mut self, code: FatalCode, detail: String) { + if self.fatal.is_some() { + return; + } + self.fatal = Some(FatalState { code, detail }); + self.checkpoint_ok = false; + } + + fn compute_stale(&self, now: Instant) -> bool { + if !self.initialized || !self.checkpoint_ok { + return false; + } + let baseline = match self.last_ingress.or(self.checkpoint_at) { + Some(t) => t, + None => return false, + }; + now.saturating_duration_since(baseline) >= self.limits.stale_after + } +} + +fn next_overload(currently: bool, depth: u64, capacity: u64, high: f64, low: f64) -> bool { + if capacity == 0 { + return false; + } + let ratio = depth as f64 / capacity as f64; + if currently { + ratio > low + } else { + ratio >= high + } +} + +fn duration_ms(d: Duration) -> u64 { + u64::try_from(d.as_millis()).unwrap_or(u64::MAX) +} + +/// Cloneable, non-tick-blocking handle for supervisors and the control surface. +#[derive(Clone)] +pub struct HealthHandle { + inner: Arc>, +} + +impl HealthHandle { + /// Live process, not yet ready. Used when the daemon is constructed. + pub fn started(limits: HealthLimits) -> Self { + let mut machine = HealthMachine::new(SystemClock, limits); + machine.apply(HealthEvent::ProcessStarted); + Self { + inner: Arc::new(RwLock::new(machine)), + } + } + + pub fn from_machine(machine: HealthMachine) -> Self { + Self { + inner: Arc::new(RwLock::new(machine)), + } + } + + pub fn apply(&self, event: HealthEvent) { + let mut guard = self.inner.write().unwrap_or_else(|e| e.into_inner()); + guard.apply(event); + } + + /// Clone the current snapshot. May wait only for an in-flight `apply` (no I/O). + pub fn snapshot(&self) -> HealthSnapshot { + let guard = self.inner.read().unwrap_or_else(|e| e.into_inner()); + guard.snapshot() + } + + /// Never waits. Returns `None` if a writer currently holds the lock. + pub fn try_snapshot(&self) -> Option { + self.inner.try_read().ok().map(|guard| guard.snapshot()) + } + + #[cfg(test)] + pub(crate) fn lock_write_for_test(&self) -> std::sync::RwLockWriteGuard<'_, HealthMachine> { + self.inner.write().unwrap_or_else(|e| e.into_inner()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::time::Duration; + + fn limits() -> HealthLimits { + HealthLimits { + stale_after: Duration::from_millis(100), + overload_high: 0.90, + overload_low: 0.70, + } + } + + fn machine() -> (HealthMachine, FakeClock) { + let clock = FakeClock::new(); + let machine = HealthMachine::new(clock.clone(), limits()); + (machine, clock) + } + + fn ckpt() -> CheckpointIdentity { + CheckpointIdentity { + id: "soma16".into(), + digest: Some("abc123".into()), + } + } + + fn bring_ready(m: &mut HealthMachine) { + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::IngressObserved); + m.apply(HealthEvent::TickSucceeded); + } + + #[test] + fn process_start_is_live_but_not_ready() { + let (mut m, _) = machine(); + let before = m.snapshot(); + assert!(!before.live); + assert!(!before.ready); + assert_eq!(before.phase, HealthPhase::Starting); + + m.apply(HealthEvent::ProcessStarted); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Starting); + assert_eq!(snap.reasons, vec![ReasonCode::Starting]); + } + + #[test] + fn initialization_does_not_imply_readiness() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::LoadingCheckpoint); + assert_eq!(snap.reasons, vec![ReasonCode::CheckpointPending]); + } + + #[test] + fn checkpoint_validation_makes_ready() { + let (mut m, _) = machine(); + bring_ready(&mut m); + let snap = m.snapshot(); + assert!(snap.live); + assert!(snap.ready); + assert_eq!(snap.phase, HealthPhase::Running); + assert!(snap.reasons.is_empty()); + assert_eq!( + snap.checkpoint.as_ref().map(|c| c.id.as_str()), + Some("soma16") + ); + assert_eq!(snap.last_successful_tick_ms, Some(0)); + assert!(!snap.input_freshness.stale); + } + + #[test] + fn checkpoint_before_init_is_ignored() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + let snap = m.snapshot(); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Starting); + assert!(snap.checkpoint.is_none()); + } + + #[test] + fn stale_input_degrades_and_recovers_only_after_fresh_ingress() { + let (mut m, clock) = machine(); + bring_ready(&mut m); + + clock.advance(Duration::from_millis(99)); + let still_ok = m.snapshot(); + assert_eq!(still_ok.phase, HealthPhase::Running); + assert!(still_ok.ready); + assert!(!still_ok.input_freshness.stale); + + clock.advance(Duration::from_millis(1)); + let degraded = m.snapshot(); + assert!(degraded.live); + assert!( + degraded.ready, + "stale input is recoverable degradation, not unreadiness" + ); + assert_eq!(degraded.phase, HealthPhase::Degraded); + assert_eq!(degraded.reasons, vec![ReasonCode::StaleInput]); + assert!(degraded.input_freshness.stale); + + clock.advance(Duration::from_millis(50)); + m.apply(HealthEvent::TickSucceeded); + let still_stale = m.snapshot(); + assert!( + still_stale.input_freshness.stale, + "ticks without ingress must not clear stale_input" + ); + + m.apply(HealthEvent::IngressObserved); + let recovered = m.snapshot(); + assert_eq!(recovered.phase, HealthPhase::Running); + assert!(!recovered.input_freshness.stale); + assert!(recovered.reasons.is_empty()); + assert!(recovered.ready); + } + + #[test] + fn overload_uses_hysteresis_and_clears_only_below_low_watermark() { + let (mut m, _) = machine(); + bring_ready(&mut m); + + m.apply(HealthEvent::QueuePressure { + depth: 89, + capacity: 100, + }); + assert_eq!(m.snapshot().phase, HealthPhase::Running); + + m.apply(HealthEvent::QueuePressure { + depth: 90, + capacity: 100, + }); + let high = m.snapshot(); + assert_eq!(high.phase, HealthPhase::Degraded); + assert!(high.ready); + assert_eq!(high.reasons, vec![ReasonCode::Overload]); + assert!(high.queue_pressure.overloaded); + + m.apply(HealthEvent::QueuePressure { + depth: 80, + capacity: 100, + }); + let mid = m.snapshot(); + assert!( + mid.queue_pressure.overloaded, + "must stay overloaded between high and low watermarks" + ); + assert_eq!(mid.phase, HealthPhase::Degraded); + + m.apply(HealthEvent::QueuePressure { + depth: 70, + capacity: 100, + }); + let recovered = m.snapshot(); + assert!(!recovered.queue_pressure.overloaded); + assert_eq!(recovered.phase, HealthPhase::Running); + assert!(recovered.ready); + } + + #[test] + fn zero_capacity_queue_is_not_overload() { + let (mut m, _) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::QueuePressure { + depth: 0, + capacity: 0, + }); + let snap = m.snapshot(); + assert!(!snap.queue_pressure.overloaded); + assert_eq!(snap.queue_pressure.ratio, None); + assert_eq!(snap.phase, HealthPhase::Running); + } + + #[test] + fn combined_degradation_clears_independently() { + let (mut m, clock) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::QueuePressure { + depth: 95, + capacity: 100, + }); + clock.advance(Duration::from_millis(100)); + let both = m.snapshot(); + assert_eq!( + both.reasons, + vec![ReasonCode::StaleInput, ReasonCode::Overload] + ); + + m.apply(HealthEvent::IngressObserved); + let only_overload = m.snapshot(); + assert_eq!(only_overload.reasons, vec![ReasonCode::Overload]); + assert_eq!(only_overload.phase, HealthPhase::Degraded); + + m.apply(HealthEvent::QueuePressure { + depth: 10, + capacity: 100, + }); + let clear = m.snapshot(); + assert!(clear.reasons.is_empty()); + assert_eq!(clear.phase, HealthPhase::Running); + } + + #[test] + fn draining_drops_readiness_and_does_not_return_to_ready() { + let (mut m, _) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::BeginDrain); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Draining); + assert!(snap.reasons.contains(&ReasonCode::Draining)); + + m.apply(HealthEvent::TickSucceeded); + m.apply(HealthEvent::IngressObserved); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::InitializationCompleted); + let after = m.snapshot(); + assert!(!after.ready); + assert_eq!(after.phase, HealthPhase::Draining); + } + + #[test] + fn initialization_failure_is_sticky_fatal() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationFailed { + detail: "socket bind exploded with secret=hunter2".into(), + }); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Fatal); + assert_eq!(snap.reasons, vec![ReasonCode::Fatal]); + assert_eq!( + snap.fatal.as_ref().map(|f| f.code), + Some(FatalCode::InitializationFailed) + ); + + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::TickSucceeded); + let after = m.snapshot(); + assert!(!after.ready); + assert_eq!(after.phase, HealthPhase::Fatal); + assert_eq!( + after.fatal.as_ref().map(|f| f.code), + Some(FatalCode::InitializationFailed) + ); + } + + #[test] + fn checkpoint_reject_is_sticky_fatal_from_loading() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointRejected { + detail: "digest mismatch".into(), + }); + assert_eq!(m.snapshot().phase, HealthPhase::Fatal); + + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + assert!(!m.snapshot().ready); + assert_eq!(m.snapshot().phase, HealthPhase::Fatal); + } + + #[test] + fn fatal_from_running_and_degraded_never_returns_to_ready() { + let (mut running, _) = machine(); + bring_ready(&mut running); + running.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "network step invariant broken".into(), + }); + assert_eq!(running.snapshot().phase, HealthPhase::Fatal); + running.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + assert!(!running.snapshot().ready); + + let (mut degraded, clock) = machine(); + bring_ready(&mut degraded); + clock.advance(Duration::from_millis(100)); + assert_eq!(degraded.snapshot().phase, HealthPhase::Degraded); + degraded.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "boom".into(), + }); + degraded.apply(HealthEvent::IngressObserved); + degraded.apply(HealthEvent::BeginDrain); + let snap = degraded.snapshot(); + assert_eq!(snap.phase, HealthPhase::Fatal); + assert!(!snap.ready); + assert!(snap.live); + } + + #[test] + fn fatal_from_draining_stays_fatal() { + let (mut m, _) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::BeginDrain); + m.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "flush failed".into(), + }); + let snap = m.snapshot(); + assert_eq!(snap.phase, HealthPhase::Fatal); + assert!(!snap.ready); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + assert_eq!(m.snapshot().phase, HealthPhase::Fatal); + } + + #[test] + fn second_fatal_does_not_replace_the_first() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationFailed { + detail: "first".into(), + }); + m.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "second".into(), + }); + let fatal = m.snapshot().fatal.expect("fatal"); + assert_eq!(fatal.code, FatalCode::InitializationFailed); + assert_eq!(fatal.detail, "first"); + } + + #[test] + fn prometheus_labels_are_low_cardinality_and_omit_detail() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationFailed { + detail: "secret token xyzzy should never be a label".into(), + }); + let text = m.snapshot().prometheus_text(); + assert!(text.contains("brainstem_live 1")); + assert!(text.contains("brainstem_ready 0")); + assert!(text.contains("brainstem_fatal 1")); + assert!(text.contains("phase=\"fatal\"")); + assert!(!text.contains("xyzzy")); + assert!(!text.contains("secret token")); + assert!(text.contains("reason=\"stale_input\"")); + assert!(text.contains("reason=\"overload\"")); + } + + #[test] + fn try_snapshot_does_not_block_on_write_lock() { + let handle = HealthHandle::started(limits()); + let start = Instant::now(); + { + let _guard = handle.lock_write_for_test(); + assert!(handle.try_snapshot().is_none()); + } + assert!(start.elapsed() < Duration::from_millis(50)); + assert!(handle.try_snapshot().is_some()); + let snap = handle.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + } + + #[test] + fn example_snapshots_match_documented_shapes() { + let (mut healthy, _) = machine(); + bring_ready(&mut healthy); + let healthy = healthy.snapshot(); + assert_eq!( + serde_json::to_value(&healthy).unwrap(), + serde_json::json!({ + "live": true, + "ready": true, + "phase": "running", + "reasons": [], + "last_successful_tick_ms": 0, + "checkpoint": { "id": "soma16", "digest": "abc123" }, + "input_freshness": { "age_ms": 0, "stale": false }, + "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, + "fatal": null, + "observed_at_ms": 0 + }) + ); + + let (mut degraded, clock) = machine(); + bring_ready(&mut degraded); + degraded.apply(HealthEvent::QueuePressure { + depth: 95, + capacity: 100, + }); + clock.advance(Duration::from_millis(100)); + let degraded = degraded.snapshot(); + assert_eq!(degraded.phase, HealthPhase::Degraded); + assert_eq!( + degraded.reasons, + vec![ReasonCode::StaleInput, ReasonCode::Overload] + ); + assert!(degraded.ready); + + let (mut fatal, _) = machine(); + fatal.apply(HealthEvent::ProcessStarted); + fatal.apply(HealthEvent::InitializationCompleted); + fatal.apply(HealthEvent::CheckpointRejected { + detail: "blank weights".into(), + }); + let fatal = fatal.snapshot(); + assert_eq!(fatal.phase, HealthPhase::Fatal); + assert!(!fatal.ready); + assert!(fatal.live); + assert_eq!(fatal.fatal.unwrap().code, FatalCode::CheckpointInvalid); + } +} diff --git a/src/lib.rs b/src/lib.rs index d6b4281..a08823d 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -4,8 +4,14 @@ //! Brainstem daemon library: config-driven service registry and runtime. pub mod backend; +pub mod control; pub mod daemon; +pub mod health; pub mod registry; // Re-export the new pluggable I/O surface (pub from day one). pub use backend::{BackendPair, IngressPacket, SpikeEvent, SpikeSink, StimulusSource}; +pub use health::{ + CheckpointIdentity, FakeClock, FatalCode, HealthEvent, HealthHandle, HealthLimits, + HealthMachine, HealthPhase, HealthSnapshot, ReasonCode, SystemClock, +}; From fff74c5e771edebb26078835ece57a661626e99d Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 05:10:12 +0000 Subject: [PATCH 02/18] fix: drop duplicate #[test] on pre-run health assertion Clippy -D warnings rejected the duplicated attribute on the daemon_is_live_not_ready_before_run test. Co-authored-by: Raul Cardenas Montoya --- src/daemon.rs | 1 - 1 file changed, 1 deletion(-) diff --git a/src/daemon.rs b/src/daemon.rs index efec49c..29e3c98 100644 --- a/src/daemon.rs +++ b/src/daemon.rs @@ -507,7 +507,6 @@ mod tests { assert!(daemon.registry().contains("critic-ipc")); } - #[test] #[test] fn daemon_is_live_not_ready_before_run() { let daemon = BrainstemDaemon::new(sample_config()); From 1c8ae3635e7d2f985f816799ac655528cabfae61 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 05:43:16 +0000 Subject: [PATCH 03/18] fix: address DeepSource/Codacy health-surface findings Invert FakeClock Default/new to avoid RS-A1008, prefer Default for empty constructors (RS-W1079), and split the health module to cut file complexity. Also bind the control listener before spawn, let run() own initialize, bound control I/O, serialize digest as null, and only mark ingress when a packet carries data. Co-authored-by: Raul Cardenas Montoya --- README.md | 2 +- docs/health.md | 10 +- src/bin/brainstem_daemon.rs | 13 +- src/control.rs | 83 ++- src/daemon.rs | 68 ++- src/health.rs | 983 ------------------------------------ src/health/mod.rs | 614 ++++++++++++++++++++++ src/health/tests.rs | 450 +++++++++++++++++ 8 files changed, 1179 insertions(+), 1044 deletions(-) delete mode 100644 src/health.rs create mode 100644 src/health/mod.rs create mode 100644 src/health/tests.rs diff --git a/README.md b/README.md index c8de7c7..fb6ed08 100644 --- a/README.md +++ b/README.md @@ -107,7 +107,7 @@ Default Cargo features are empty (`default = []` in `Cargo.toml`). That path use Enabling the feature does **not** change `BrainstemDaemon::new()` or `try_new()`. Those always inject `BackendPair::stub()`. Only `src/bin/brainstem_daemon.rs` constructs `ZmqStimulusSource` + `ZmqSpikeSink` when `corpus-ipc` is on. -Library users who want live ZMQ must build that pair themselves under `#[cfg(feature = "corpus-ipc")]` and pass it to `with_backend` / `try_with_backend`. Call `StimulusSource::initialize(...)` on the source first (as the binary does). `run` also calls `initialize` (idempotent on success) so readiness can move past the checkpoint gate. Skipping initialize before `run` is therefore no longer required for the stub path; a failing `initialize` marks health **fatal** and never becomes ready. +Library users who want live ZMQ must build that pair themselves under `#[cfg(feature = "corpus-ipc")]` and pass it to `with_backend` / `try_with_backend`. `BrainstemDaemon::run` is the sole caller of `StimulusSource::initialize` (the binary no longer initializes first, so the pinned ZMQ backend is not reconnected). A failing `initialize` marks health **fatal** and never becomes ready. Health snapshots, probe paths, and the transition table live in [`docs/health.md`](docs/health.md). diff --git a/docs/health.md b/docs/health.md index f0e40ed..1179b5b 100644 --- a/docs/health.md +++ b/docs/health.md @@ -53,7 +53,8 @@ file name (not the full path) with `digest: null`. ## Example snapshots -Healthy (ready to consume events): +Healthy (ready to consume events). Until LIM-1133 the live daemon stand-in uses +`"digest": null`. ```json { @@ -62,7 +63,8 @@ Healthy (ready to consume events): "phase": "running", "reasons": [], "last_successful_tick_ms": 0, - "checkpoint": { "id": "soma16", "digest": "abc123" }, + "tick_age_ms": 0, + "checkpoint": { "id": "soma16", "digest": null }, "input_freshness": { "age_ms": 0, "stale": false }, "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, "fatal": null, @@ -79,7 +81,8 @@ Degraded (still ready; supervisors should not bounce the process): "phase": "degraded", "reasons": ["stale_input", "overload"], "last_successful_tick_ms": 0, - "checkpoint": { "id": "soma16", "digest": "abc123" }, + "tick_age_ms": 0, + "checkpoint": { "id": "soma16", "digest": null }, "input_freshness": { "age_ms": 100, "stale": true }, "queue_pressure": { "depth": 95, "capacity": 100, "ratio": 0.95, "overloaded": true }, "fatal": null, @@ -96,6 +99,7 @@ Fatal (never returns to ready in this process): "phase": "fatal", "reasons": ["fatal"], "last_successful_tick_ms": null, + "tick_age_ms": null, "checkpoint": null, "input_freshness": { "age_ms": null, "stale": false }, "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, diff --git a/src/bin/brainstem_daemon.rs b/src/bin/brainstem_daemon.rs index 91a83a4..6f4a5de 100644 --- a/src/bin/brainstem_daemon.rs +++ b/src/bin/brainstem_daemon.rs @@ -12,8 +12,6 @@ use brainstem_daemon::daemon::{BrainstemDaemon, DaemonConfig}; use brainstem_daemon::daemon::CORPUS_IPC_READOUT_ENV; use anyhow::Context; -#[cfg(feature = "corpus-ipc")] -use brainstem_daemon::StimulusSource; use clap::Parser; use tracing::info; use tracing_subscriber::EnvFilter; @@ -77,15 +75,10 @@ async fn run(cfg: DaemonConfig, config_path: PathBuf) -> anyhow::Result<()> { // Build a real ZMQ pair (binary is responsible for the SUB endpoint via env). // We still need to create the PUB side here because the default `new()` path // is intentionally conservative. - let mut source = brainstem_daemon::backend::ZmqStimulusSource::with_channels(cfg.channels); - - // Pass the model path through (was dropped before). The pinned - // ZmqBrainBackend::initialize takes `_model_path` and currently ignores it. + let source = brainstem_daemon::backend::ZmqStimulusSource::with_channels(cfg.channels); - let model_path = cfg.model_path.to_string_lossy(); - source - .initialize(Some(model_path.as_ref())) - .map_err(|e| anyhow::anyhow!("failed to initialize ZMQ stimulus source: {e}"))?; + // `BrainstemDaemon::run` owns `StimulusSource::initialize` so the pinned + // ZMQ backend is not reconnected here (repeat initialize replaces the SUB socket). let zmq_context = zmq::Context::new(); let pub_socket = zmq_context diff --git a/src/control.rs b/src/control.rs index de095ba..ecdaafa 100644 --- a/src/control.rs +++ b/src/control.rs @@ -8,16 +8,20 @@ //! a second server alongside this one. use std::net::SocketAddr; +use std::sync::Arc; use std::time::Duration; use anyhow::{Context, Result}; use tokio::io::{AsyncReadExt, AsyncWriteExt}; use tokio::net::{TcpListener, TcpStream}; -use tokio::sync::watch; +use tokio::sync::{Semaphore, watch}; use tracing::{info, warn}; use crate::health::{HealthHandle, HealthSnapshot}; +const IO_TIMEOUT: Duration = Duration::from_secs(2); +const MAX_CONTROL_CONNS: usize = 32; + pub async fn serve( addr: SocketAddr, health: HealthHandle, @@ -39,6 +43,11 @@ pub async fn serve_listener( .context("control listener has no local address")?; info!(%bound, "control surface listening (/livez /readyz /health /metrics)"); + if *shutdown.borrow() { + return Ok(()); + } + + let slots = Arc::new(Semaphore::new(MAX_CONTROL_CONNS)); loop { tokio::select! { changed = shutdown.changed() => { @@ -49,8 +58,13 @@ pub async fn serve_listener( accepted = listener.accept() => { match accepted { Ok((stream, _)) => { + let Ok(permit) = slots.clone().try_acquire_owned() else { + drop(stream); + continue; + }; let health = health.clone(); tokio::spawn(async move { + let _permit = permit; if let Err(e) = handle_connection(stream, &health).await { warn!("control connection failed: {e}"); } @@ -67,36 +81,43 @@ pub async fn serve_listener( async fn handle_connection(mut stream: TcpStream, health: &HealthHandle) -> Result<()> { let mut buf = [0u8; 1024]; - let n = tokio::time::timeout(Duration::from_secs(2), stream.read(&mut buf)) + let n = tokio::time::timeout(IO_TIMEOUT, read_request_line(&mut stream, &mut buf)) .await .context("control read timed out")? .context("control read failed")?; let req = std::str::from_utf8(&buf[..n]).unwrap_or(""); let snap = health.snapshot(); let response = render_http(req, &snap); - stream.write_all(&response).await?; - stream.flush().await?; + tokio::time::timeout(IO_TIMEOUT, async { + stream.write_all(&response).await?; + stream.flush().await?; + Ok::<_, std::io::Error>(()) + }) + .await + .context("control write timed out")? + .context("control write failed")?; Ok(()) } +async fn read_request_line(stream: &mut TcpStream, buf: &mut [u8]) -> std::io::Result { + let mut total = 0; + while total < buf.len() { + let n = stream.read(&mut buf[total..]).await?; + if n == 0 { + break; + } + total += n; + if buf[..total].contains(&b'\n') { + break; + } + } + Ok(total) +} + pub(crate) fn render_http(request: &str, snap: &HealthSnapshot) -> Vec { match parse_get_path(request) { ParseResult::Get(path) => { - let (status, content_type, body) = match path { - "/livez" | "/healthz/live" => probe(snap.live, b"live\n", b"not live\n"), - "/readyz" | "/healthz/ready" => probe(snap.ready, b"ready\n", b"not ready\n"), - "/health" => ( - 200, - "application/json", - serde_json::to_vec(snap).unwrap_or_else(|_| b"{}".to_vec()), - ), - "/metrics" => ( - 200, - "text/plain; version=0.0.4", - snap.prometheus_text().into_bytes(), - ), - _ => (404, "text/plain; charset=utf-8", b"not found\n".to_vec()), - }; + let (status, content_type, body) = route(path, snap); http_response(status, content_type, &body) } ParseResult::NotGet => { @@ -106,6 +127,24 @@ pub(crate) fn render_http(request: &str, snap: &HealthSnapshot) -> Vec { } } +fn route(path: &str, snap: &HealthSnapshot) -> (u16, &'static str, Vec) { + match path { + "/livez" | "/healthz/live" => probe(snap.live, b"live\n", b"not live\n"), + "/readyz" | "/healthz/ready" => probe(snap.ready, b"ready\n", b"not ready\n"), + "/health" => ( + 200, + "application/json", + serde_json::to_vec(snap).unwrap_or_else(|_| b"{}".to_vec()), + ), + "/metrics" => ( + 200, + "text/plain; version=0.0.4", + snap.prometheus_text().into_bytes(), + ), + _ => (404, "text/plain; charset=utf-8", b"not found\n".to_vec()), + } +} + fn probe(ok: bool, yes: &'static [u8], no: &'static [u8]) -> (u16, &'static str, Vec) { if ok { (200, "text/plain; charset=utf-8", yes.to_vec()) @@ -171,7 +210,7 @@ mod tests { }; fn ready_snap() -> HealthSnapshot { - let clock = FakeClock::new(); + let clock = FakeClock::default(); let mut machine = HealthMachine::new(clock, HealthLimits::default()); machine.apply(HealthEvent::ProcessStarted); machine.apply(HealthEvent::InitializationCompleted); @@ -193,7 +232,7 @@ mod tests { let ready = String::from_utf8(render_http("GET /readyz HTTP/1.1\r\n\r\n", &snap)).unwrap(); assert!(ready.starts_with("HTTP/1.1 200 OK")); - let starting = HealthMachine::new(FakeClock::new(), HealthLimits::default()); + let starting = HealthMachine::new(FakeClock::default(), HealthLimits::default()); let mut starting_m = starting; starting_m.apply(HealthEvent::ProcessStarted); let starting = starting_m.snapshot(); @@ -209,7 +248,7 @@ mod tests { #[test] fn health_json_is_always_200() { - let mut machine = HealthMachine::new(FakeClock::new(), HealthLimits::default()); + let mut machine = HealthMachine::new(FakeClock::default(), HealthLimits::default()); machine.apply(HealthEvent::ProcessStarted); let snap = machine.snapshot(); let raw = String::from_utf8(render_http("GET /health HTTP/1.1\r\n\r\n", &snap)).unwrap(); diff --git a/src/daemon.rs b/src/daemon.rs index 29e3c98..e2591a0 100644 --- a/src/daemon.rs +++ b/src/daemon.rs @@ -160,26 +160,10 @@ impl BrainstemDaemon { let (control_stop, control_task) = start_control(cfg.control_bind.as_deref(), health.clone()).await?; - let model_path = cfg.model_path.to_string_lossy(); - if let Err(e) = backend.source.initialize(Some(model_path.as_ref())) { - health.apply(HealthEvent::InitializationFailed { - detail: e.to_string(), - }); - error!("Stimulus source initialization failed: {e}"); - await_shutdown_if_control(control_stop.is_some()).await; + if let Err(e) = initialize_source(&mut *backend.source, &cfg, &health) { stop_control(control_stop, control_task).await; - return Err(e).context("failed to initialize stimulus source"); + return Err(e); } - health.apply(HealthEvent::InitializationCompleted); - - // Successful initialize is the current checkpoint gate. Real digest/weight - // validation is tracked in LIM-1133 and will replace this stand-in. - health.apply(HealthEvent::CheckpointValidated { - identity: CheckpointIdentity { - id: checkpoint_id_from_path(&cfg.model_path), - digest: None, - }, - }); let tick_duration = Duration::from_nanos(1_000_000_000 / u64::from(cfg.tick_rate_hz)); let mut ticker = time::interval(tick_duration); @@ -286,9 +270,12 @@ async fn start_control( let addr: SocketAddr = bind .parse() .with_context(|| format!("invalid control_bind {bind}"))?; + let listener = tokio::net::TcpListener::bind(addr) + .await + .with_context(|| format!("failed to bind control surface on {addr}"))?; let (tx, rx) = watch::channel(false); let task = tokio::spawn(async move { - if let Err(e) = crate::control::serve(addr, health, rx).await { + if let Err(e) = crate::control::serve_listener(listener, health, rx).await { warn!("control surface stopped: {e}"); } }); @@ -307,10 +294,29 @@ async fn stop_control( } } -async fn await_shutdown_if_control(has_control: bool) { - if has_control { - shutdown_signal().await; - } +fn initialize_source( + source: &mut dyn StimulusSource, + cfg: &DaemonConfig, + health: &HealthHandle, +) -> Result<()> { + let model_path = cfg.model_path.to_string_lossy(); + if let Err(e) = source.initialize(Some(model_path.as_ref())) { + health.apply(HealthEvent::InitializationFailed { + detail: "stimulus source initialization failed".into(), + }); + error!("Stimulus source initialization failed: {e}"); + return Err(e).context("failed to initialize stimulus source"); + } + health.apply(HealthEvent::InitializationCompleted); + // Successful initialize is the current checkpoint gate. Real digest/weight + // validation is tracked in LIM-1133 and will replace this stand-in. + health.apply(HealthEvent::CheckpointValidated { + identity: CheckpointIdentity { + id: checkpoint_id_from_path(&cfg.model_path), + digest: None, + }, + }); + Ok(()) } fn checkpoint_id_from_path(path: &Path) -> String { @@ -321,6 +327,14 @@ fn checkpoint_id_from_path(path: &Path) -> String { .to_string() } +fn packet_carries_input(packet: &IngressPacket) -> bool { + !packet.stimuli.is_empty() + || packet + .modulators + .as_ref() + .is_some_and(|mods| !mods.is_empty()) +} + fn validate_neuron_count(config: &DaemonConfig) -> Result<()> { let total = config .lif_count @@ -357,7 +371,9 @@ fn run_tick( ) { let packet = match source.next_ingress() { Ok(Some(p)) => { - health.apply(HealthEvent::IngressObserved); + if packet_carries_input(&p) { + health.apply(HealthEvent::IngressObserved); + } p } Ok(None) => { @@ -386,7 +402,6 @@ fn run_tick( return; } }; - health.apply(HealthEvent::TickSucceeded); // Single timestamp for both per-spike time and batch metadata (keeps them consistent). let now = SystemTime::now() @@ -420,6 +435,7 @@ fn run_tick( if spike_buf.is_empty() && !spike_ids.is_empty() { // Had spikes from network but all IDs were out of u16 range (dropped). // Nothing valid to publish; skip to avoid empty batch for dropped case. + health.apply(HealthEvent::TickSucceeded); return; } @@ -430,7 +446,9 @@ fn run_tick( // expectations (CollectingSpikeSink) and wire behavior stable. if let Err(e) = sink.emit(spike_buf, now) { warn!("Failed to emit spikes: {e}"); + return; } + health.apply(HealthEvent::TickSucceeded); } /// decode_inputs now takes an IngressPacket. diff --git a/src/health.rs b/src/health.rs deleted file mode 100644 index 0bdaf00..0000000 --- a/src/health.rs +++ /dev/null @@ -1,983 +0,0 @@ -// SPDX-License-Identifier: MIT OR Apache-2.0 -// Copyright 2026 Raul Montoya Cardenas - -//! Process health: liveness, readiness, recoverable degradation, and sticky fatal state. -//! -//! Supervisors should treat [`HealthSnapshot::live`] and [`HealthSnapshot::ready`] as -//! independent. A live process has not necessarily loaded a valid checkpoint. Recoverable -//! reasons (`stale_input`, `overload`) clear only after the condition is observed healthy. -//! Fatal and draining states never return to ready in the same process. -//! -//! Snapshot reads take a separate lock from the tick loop's backend and network, so they -//! do not wait on `StimulusSource` or `SpikingNetwork::step`. - -use std::sync::atomic::{AtomicU64, Ordering}; -use std::sync::{Arc, RwLock}; -use std::time::{Duration, Instant}; - -use serde::{Deserialize, Serialize}; - -/// Monotonic clock used by the health state machine. -pub trait Clock: Send + Sync { - fn now(&self) -> Instant; -} - -/// Wall-clock monotonic clock. -#[derive(Debug, Clone, Copy, Default)] -pub struct SystemClock; - -impl Clock for SystemClock { - fn now(&self) -> Instant { - Instant::now() - } -} - -/// Test clock. Clone and share the same offset with a [`HealthMachine`]. -#[derive(Debug, Clone)] -pub struct FakeClock { - origin: Instant, - offset_nanos: Arc, -} - -impl FakeClock { - pub fn new() -> Self { - Self { - origin: Instant::now(), - offset_nanos: Arc::new(AtomicU64::new(0)), - } - } - - pub fn advance(&self, duration: Duration) { - let add = u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX); - self.offset_nanos.fetch_add(add, Ordering::SeqCst); - } -} - -impl Default for FakeClock { - fn default() -> Self { - Self::new() - } -} - -impl Clock for FakeClock { - fn now(&self) -> Instant { - self.origin + Duration::from_nanos(self.offset_nanos.load(Ordering::SeqCst)) - } -} - -/// Thresholds for recoverable degradation. -#[derive(Debug, Clone, Copy, PartialEq)] -pub struct HealthLimits { - /// Ingress older than this marks `stale_input`. - pub stale_after: Duration, - /// Queue fill ratio that *enters* overload (`depth / capacity`). - pub overload_high: f64, - /// Queue fill ratio that *exits* overload (hysteresis; must be `<= overload_high`). - pub overload_low: f64, -} - -impl Default for HealthLimits { - fn default() -> Self { - Self { - stale_after: Duration::from_millis(500), - overload_high: 0.90, - overload_low: 0.70, - } - } -} - -/// Coarse phase derived from the snapshot. Stable, low-cardinality. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum HealthPhase { - Starting, - LoadingCheckpoint, - Running, - Degraded, - Draining, - Fatal, -} - -impl HealthPhase { - pub const ALL: [HealthPhase; 6] = [ - HealthPhase::Starting, - HealthPhase::LoadingCheckpoint, - HealthPhase::Running, - HealthPhase::Degraded, - HealthPhase::Draining, - HealthPhase::Fatal, - ]; - - pub fn as_str(self) -> &'static str { - match self { - HealthPhase::Starting => "starting", - HealthPhase::LoadingCheckpoint => "loading_checkpoint", - HealthPhase::Running => "running", - HealthPhase::Degraded => "degraded", - HealthPhase::Draining => "draining", - HealthPhase::Fatal => "fatal", - } - } -} - -/// Stable reason codes. Never put detailed error text in metric labels — use these codes. -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum ReasonCode { - Starting, - CheckpointPending, - StaleInput, - Overload, - Draining, - Fatal, -} - -impl ReasonCode { - pub const RECOVERABLE: [ReasonCode; 2] = [ReasonCode::StaleInput, ReasonCode::Overload]; - - pub fn as_str(self) -> &'static str { - match self { - ReasonCode::Starting => "starting", - ReasonCode::CheckpointPending => "checkpoint_pending", - ReasonCode::StaleInput => "stale_input", - ReasonCode::Overload => "overload", - ReasonCode::Draining => "draining", - ReasonCode::Fatal => "fatal", - } - } -} - -/// Sticky fatal class. Low-cardinality; details live on [`FatalState::detail`]. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum FatalCode { - InitializationFailed, - CheckpointInvalid, - Unspecified, -} - -impl FatalCode { - pub fn as_str(self) -> &'static str { - match self { - FatalCode::InitializationFailed => "initialization_failed", - FatalCode::CheckpointInvalid => "checkpoint_invalid", - FatalCode::Unspecified => "unspecified", - } - } -} - -/// Identity of the loaded checkpoint. Digest may be absent until real validation lands. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct CheckpointIdentity { - pub id: String, - #[serde(skip_serializing_if = "Option::is_none")] - pub digest: Option, -} - -/// Fatal snapshot payload. `detail` is for logs/JSON, never a Prometheus label. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct FatalState { - pub code: FatalCode, - pub detail: String, -} - -/// Ingress freshness relative to the health clock. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct InputFreshness { - pub age_ms: Option, - pub stale: bool, -} - -/// Bounded-queue pressure. `capacity == 0` means "no queue instrumented". -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct QueuePressure { - pub depth: u64, - pub capacity: u64, - pub ratio: Option, - pub overloaded: bool, -} - -/// Machine-readable health view for supervisors and the control surface. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct HealthSnapshot { - pub live: bool, - pub ready: bool, - pub phase: HealthPhase, - pub reasons: Vec, - pub last_successful_tick_ms: Option, - pub checkpoint: Option, - pub input_freshness: InputFreshness, - pub queue_pressure: QueuePressure, - pub fatal: Option, - pub observed_at_ms: u64, -} - -impl HealthSnapshot { - /// Prometheus text exposition. Labels are stable reason/phase codes only. - pub fn prometheus_text(&self) -> String { - let mut out = String::new(); - out.push_str("# HELP brainstem_live 1 if the process health reporter is running.\n"); - out.push_str("# TYPE brainstem_live gauge\n"); - out.push_str(&format!("brainstem_live {}\n", u8::from(self.live))); - - out.push_str( - "# HELP brainstem_ready 1 if initialized, checkpoint-valid, not draining, not fatal.\n", - ); - out.push_str("# TYPE brainstem_ready gauge\n"); - out.push_str(&format!("brainstem_ready {}\n", u8::from(self.ready))); - - out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); - out.push_str("# TYPE brainstem_phase gauge\n"); - for phase in HealthPhase::ALL { - out.push_str(&format!( - "brainstem_phase{{phase=\"{}\"}} {}\n", - phase.as_str(), - u8::from(self.phase == phase) - )); - } - - out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); - out.push_str("# TYPE brainstem_degraded gauge\n"); - for reason in ReasonCode::RECOVERABLE { - let active = self.reasons.contains(&reason); - out.push_str(&format!( - "brainstem_degraded{{reason=\"{}\"}} {}\n", - reason.as_str(), - u8::from(active) - )); - } - - out.push_str( - "# HELP brainstem_fatal 1 if this process has entered a sticky fatal state.\n", - ); - out.push_str("# TYPE brainstem_fatal gauge\n"); - out.push_str(&format!( - "brainstem_fatal {}\n", - u8::from(self.fatal.is_some()) - )); - - out.push_str( - "# HELP brainstem_last_successful_tick_ms Milliseconds since start of last successful tick.\n", - ); - out.push_str("# TYPE brainstem_last_successful_tick_ms gauge\n"); - out.push_str(&format!( - "brainstem_last_successful_tick_ms {}\n", - self.last_successful_tick_ms.unwrap_or(0) - )); - - out.push_str("# HELP brainstem_input_age_ms Age of last ingress sample in milliseconds.\n"); - out.push_str("# TYPE brainstem_input_age_ms gauge\n"); - out.push_str(&format!( - "brainstem_input_age_ms {}\n", - self.input_freshness.age_ms.unwrap_or(0) - )); - - out.push_str("# HELP brainstem_queue_depth Ingress queue depth.\n"); - out.push_str("# TYPE brainstem_queue_depth gauge\n"); - out.push_str(&format!( - "brainstem_queue_depth {}\n", - self.queue_pressure.depth - )); - - out.push_str("# HELP brainstem_queue_capacity Ingress queue capacity.\n"); - out.push_str("# TYPE brainstem_queue_capacity gauge\n"); - out.push_str(&format!( - "brainstem_queue_capacity {}\n", - self.queue_pressure.capacity - )); - - out - } -} - -/// State-machine events. Tick-loop I/O never runs while these are applied. -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum HealthEvent { - ProcessStarted, - InitializationCompleted, - InitializationFailed { detail: String }, - CheckpointValidated { identity: CheckpointIdentity }, - CheckpointRejected { detail: String }, - TickSucceeded, - IngressObserved, - QueuePressure { depth: u64, capacity: u64 }, - BeginDrain, - Fatal { code: FatalCode, detail: String }, -} - -/// Pure health state machine. Drive it with [`FakeClock`] in tests. -pub struct HealthMachine { - clock: Arc, - limits: HealthLimits, - started: bool, - initialized: bool, - checkpoint_ok: bool, - draining: bool, - started_at: Option, - checkpoint_at: Option, - last_tick: Option, - last_ingress: Option, - checkpoint: Option, - queue_depth: u64, - queue_capacity: u64, - overloaded: bool, - fatal: Option, -} - -impl HealthMachine { - pub fn new(clock: impl Clock + 'static, limits: HealthLimits) -> Self { - Self { - clock: Arc::new(clock), - limits, - started: false, - initialized: false, - checkpoint_ok: false, - draining: false, - started_at: None, - checkpoint_at: None, - last_tick: None, - last_ingress: None, - checkpoint: None, - queue_depth: 0, - queue_capacity: 0, - overloaded: false, - fatal: None, - } - } - - pub fn apply(&mut self, event: HealthEvent) { - let now = self.clock.now(); - match event { - HealthEvent::ProcessStarted => { - if !self.started { - self.started = true; - self.started_at = Some(now); - } - } - HealthEvent::InitializationCompleted => { - if self.fatal.is_some() || self.draining { - return; - } - self.initialized = true; - } - HealthEvent::InitializationFailed { detail } => { - self.enter_fatal(FatalCode::InitializationFailed, detail); - } - HealthEvent::CheckpointValidated { identity } => { - if self.fatal.is_some() || self.draining { - return; - } - if !self.initialized { - return; - } - self.checkpoint = Some(identity); - self.checkpoint_ok = true; - self.checkpoint_at = Some(now); - } - HealthEvent::CheckpointRejected { detail } => { - self.enter_fatal(FatalCode::CheckpointInvalid, detail); - } - HealthEvent::TickSucceeded => { - self.last_tick = Some(now); - } - HealthEvent::IngressObserved => { - self.last_ingress = Some(now); - } - HealthEvent::QueuePressure { depth, capacity } => { - self.queue_depth = depth; - self.queue_capacity = capacity; - self.overloaded = next_overload( - self.overloaded, - depth, - capacity, - self.limits.overload_high, - self.limits.overload_low, - ); - } - HealthEvent::BeginDrain => { - if self.fatal.is_some() { - return; - } - self.draining = true; - } - HealthEvent::Fatal { code, detail } => { - self.enter_fatal(code, detail); - } - } - } - - pub fn snapshot(&self) -> HealthSnapshot { - let now = self.clock.now(); - let origin = self.started_at.unwrap_or(now); - let ms = |t: Instant| duration_ms(t.saturating_duration_since(origin)); - let observed_at_ms = ms(now); - - let age_ms = self - .last_ingress - .map(|t| duration_ms(now.saturating_duration_since(t))); - let stale = self.compute_stale(now); - let live = self.started; - let ready = live - && self.initialized - && self.checkpoint_ok - && !self.draining - && self.fatal.is_none(); - - let mut reasons = Vec::new(); - if self.fatal.is_some() { - reasons.push(ReasonCode::Fatal); - } else { - if !self.initialized { - reasons.push(ReasonCode::Starting); - } else if !self.checkpoint_ok { - reasons.push(ReasonCode::CheckpointPending); - } - if self.draining { - reasons.push(ReasonCode::Draining); - } - if stale { - reasons.push(ReasonCode::StaleInput); - } - if self.overloaded { - reasons.push(ReasonCode::Overload); - } - } - - let phase = if self.fatal.is_some() { - HealthPhase::Fatal - } else if self.draining { - HealthPhase::Draining - } else if !self.initialized { - HealthPhase::Starting - } else if !self.checkpoint_ok { - HealthPhase::LoadingCheckpoint - } else if stale || self.overloaded { - HealthPhase::Degraded - } else { - HealthPhase::Running - }; - - let ratio = if self.queue_capacity == 0 { - None - } else { - Some(self.queue_depth as f64 / self.queue_capacity as f64) - }; - - HealthSnapshot { - live, - ready, - phase, - reasons, - last_successful_tick_ms: self.last_tick.map(ms), - checkpoint: self.checkpoint.clone(), - input_freshness: InputFreshness { age_ms, stale }, - queue_pressure: QueuePressure { - depth: self.queue_depth, - capacity: self.queue_capacity, - ratio, - overloaded: self.overloaded, - }, - fatal: self.fatal.clone(), - observed_at_ms, - } - } - - fn enter_fatal(&mut self, code: FatalCode, detail: String) { - if self.fatal.is_some() { - return; - } - self.fatal = Some(FatalState { code, detail }); - self.checkpoint_ok = false; - } - - fn compute_stale(&self, now: Instant) -> bool { - if !self.initialized || !self.checkpoint_ok { - return false; - } - let baseline = match self.last_ingress.or(self.checkpoint_at) { - Some(t) => t, - None => return false, - }; - now.saturating_duration_since(baseline) >= self.limits.stale_after - } -} - -fn next_overload(currently: bool, depth: u64, capacity: u64, high: f64, low: f64) -> bool { - if capacity == 0 { - return false; - } - let ratio = depth as f64 / capacity as f64; - if currently { - ratio > low - } else { - ratio >= high - } -} - -fn duration_ms(d: Duration) -> u64 { - u64::try_from(d.as_millis()).unwrap_or(u64::MAX) -} - -/// Cloneable, non-tick-blocking handle for supervisors and the control surface. -#[derive(Clone)] -pub struct HealthHandle { - inner: Arc>, -} - -impl HealthHandle { - /// Live process, not yet ready. Used when the daemon is constructed. - pub fn started(limits: HealthLimits) -> Self { - let mut machine = HealthMachine::new(SystemClock, limits); - machine.apply(HealthEvent::ProcessStarted); - Self { - inner: Arc::new(RwLock::new(machine)), - } - } - - pub fn from_machine(machine: HealthMachine) -> Self { - Self { - inner: Arc::new(RwLock::new(machine)), - } - } - - pub fn apply(&self, event: HealthEvent) { - let mut guard = self.inner.write().unwrap_or_else(|e| e.into_inner()); - guard.apply(event); - } - - /// Clone the current snapshot. May wait only for an in-flight `apply` (no I/O). - pub fn snapshot(&self) -> HealthSnapshot { - let guard = self.inner.read().unwrap_or_else(|e| e.into_inner()); - guard.snapshot() - } - - /// Never waits. Returns `None` if a writer currently holds the lock. - pub fn try_snapshot(&self) -> Option { - self.inner.try_read().ok().map(|guard| guard.snapshot()) - } - - #[cfg(test)] - pub(crate) fn lock_write_for_test(&self) -> std::sync::RwLockWriteGuard<'_, HealthMachine> { - self.inner.write().unwrap_or_else(|e| e.into_inner()) - } -} - -#[cfg(test)] -mod tests { - use super::*; - use std::time::Duration; - - fn limits() -> HealthLimits { - HealthLimits { - stale_after: Duration::from_millis(100), - overload_high: 0.90, - overload_low: 0.70, - } - } - - fn machine() -> (HealthMachine, FakeClock) { - let clock = FakeClock::new(); - let machine = HealthMachine::new(clock.clone(), limits()); - (machine, clock) - } - - fn ckpt() -> CheckpointIdentity { - CheckpointIdentity { - id: "soma16".into(), - digest: Some("abc123".into()), - } - } - - fn bring_ready(m: &mut HealthMachine) { - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::InitializationCompleted); - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - m.apply(HealthEvent::IngressObserved); - m.apply(HealthEvent::TickSucceeded); - } - - #[test] - fn process_start_is_live_but_not_ready() { - let (mut m, _) = machine(); - let before = m.snapshot(); - assert!(!before.live); - assert!(!before.ready); - assert_eq!(before.phase, HealthPhase::Starting); - - m.apply(HealthEvent::ProcessStarted); - let snap = m.snapshot(); - assert!(snap.live); - assert!(!snap.ready); - assert_eq!(snap.phase, HealthPhase::Starting); - assert_eq!(snap.reasons, vec![ReasonCode::Starting]); - } - - #[test] - fn initialization_does_not_imply_readiness() { - let (mut m, _) = machine(); - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::InitializationCompleted); - let snap = m.snapshot(); - assert!(snap.live); - assert!(!snap.ready); - assert_eq!(snap.phase, HealthPhase::LoadingCheckpoint); - assert_eq!(snap.reasons, vec![ReasonCode::CheckpointPending]); - } - - #[test] - fn checkpoint_validation_makes_ready() { - let (mut m, _) = machine(); - bring_ready(&mut m); - let snap = m.snapshot(); - assert!(snap.live); - assert!(snap.ready); - assert_eq!(snap.phase, HealthPhase::Running); - assert!(snap.reasons.is_empty()); - assert_eq!( - snap.checkpoint.as_ref().map(|c| c.id.as_str()), - Some("soma16") - ); - assert_eq!(snap.last_successful_tick_ms, Some(0)); - assert!(!snap.input_freshness.stale); - } - - #[test] - fn checkpoint_before_init_is_ignored() { - let (mut m, _) = machine(); - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - let snap = m.snapshot(); - assert!(!snap.ready); - assert_eq!(snap.phase, HealthPhase::Starting); - assert!(snap.checkpoint.is_none()); - } - - #[test] - fn stale_input_degrades_and_recovers_only_after_fresh_ingress() { - let (mut m, clock) = machine(); - bring_ready(&mut m); - - clock.advance(Duration::from_millis(99)); - let still_ok = m.snapshot(); - assert_eq!(still_ok.phase, HealthPhase::Running); - assert!(still_ok.ready); - assert!(!still_ok.input_freshness.stale); - - clock.advance(Duration::from_millis(1)); - let degraded = m.snapshot(); - assert!(degraded.live); - assert!( - degraded.ready, - "stale input is recoverable degradation, not unreadiness" - ); - assert_eq!(degraded.phase, HealthPhase::Degraded); - assert_eq!(degraded.reasons, vec![ReasonCode::StaleInput]); - assert!(degraded.input_freshness.stale); - - clock.advance(Duration::from_millis(50)); - m.apply(HealthEvent::TickSucceeded); - let still_stale = m.snapshot(); - assert!( - still_stale.input_freshness.stale, - "ticks without ingress must not clear stale_input" - ); - - m.apply(HealthEvent::IngressObserved); - let recovered = m.snapshot(); - assert_eq!(recovered.phase, HealthPhase::Running); - assert!(!recovered.input_freshness.stale); - assert!(recovered.reasons.is_empty()); - assert!(recovered.ready); - } - - #[test] - fn overload_uses_hysteresis_and_clears_only_below_low_watermark() { - let (mut m, _) = machine(); - bring_ready(&mut m); - - m.apply(HealthEvent::QueuePressure { - depth: 89, - capacity: 100, - }); - assert_eq!(m.snapshot().phase, HealthPhase::Running); - - m.apply(HealthEvent::QueuePressure { - depth: 90, - capacity: 100, - }); - let high = m.snapshot(); - assert_eq!(high.phase, HealthPhase::Degraded); - assert!(high.ready); - assert_eq!(high.reasons, vec![ReasonCode::Overload]); - assert!(high.queue_pressure.overloaded); - - m.apply(HealthEvent::QueuePressure { - depth: 80, - capacity: 100, - }); - let mid = m.snapshot(); - assert!( - mid.queue_pressure.overloaded, - "must stay overloaded between high and low watermarks" - ); - assert_eq!(mid.phase, HealthPhase::Degraded); - - m.apply(HealthEvent::QueuePressure { - depth: 70, - capacity: 100, - }); - let recovered = m.snapshot(); - assert!(!recovered.queue_pressure.overloaded); - assert_eq!(recovered.phase, HealthPhase::Running); - assert!(recovered.ready); - } - - #[test] - fn zero_capacity_queue_is_not_overload() { - let (mut m, _) = machine(); - bring_ready(&mut m); - m.apply(HealthEvent::QueuePressure { - depth: 0, - capacity: 0, - }); - let snap = m.snapshot(); - assert!(!snap.queue_pressure.overloaded); - assert_eq!(snap.queue_pressure.ratio, None); - assert_eq!(snap.phase, HealthPhase::Running); - } - - #[test] - fn combined_degradation_clears_independently() { - let (mut m, clock) = machine(); - bring_ready(&mut m); - m.apply(HealthEvent::QueuePressure { - depth: 95, - capacity: 100, - }); - clock.advance(Duration::from_millis(100)); - let both = m.snapshot(); - assert_eq!( - both.reasons, - vec![ReasonCode::StaleInput, ReasonCode::Overload] - ); - - m.apply(HealthEvent::IngressObserved); - let only_overload = m.snapshot(); - assert_eq!(only_overload.reasons, vec![ReasonCode::Overload]); - assert_eq!(only_overload.phase, HealthPhase::Degraded); - - m.apply(HealthEvent::QueuePressure { - depth: 10, - capacity: 100, - }); - let clear = m.snapshot(); - assert!(clear.reasons.is_empty()); - assert_eq!(clear.phase, HealthPhase::Running); - } - - #[test] - fn draining_drops_readiness_and_does_not_return_to_ready() { - let (mut m, _) = machine(); - bring_ready(&mut m); - m.apply(HealthEvent::BeginDrain); - let snap = m.snapshot(); - assert!(snap.live); - assert!(!snap.ready); - assert_eq!(snap.phase, HealthPhase::Draining); - assert!(snap.reasons.contains(&ReasonCode::Draining)); - - m.apply(HealthEvent::TickSucceeded); - m.apply(HealthEvent::IngressObserved); - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - m.apply(HealthEvent::InitializationCompleted); - let after = m.snapshot(); - assert!(!after.ready); - assert_eq!(after.phase, HealthPhase::Draining); - } - - #[test] - fn initialization_failure_is_sticky_fatal() { - let (mut m, _) = machine(); - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::InitializationFailed { - detail: "socket bind exploded with secret=hunter2".into(), - }); - let snap = m.snapshot(); - assert!(snap.live); - assert!(!snap.ready); - assert_eq!(snap.phase, HealthPhase::Fatal); - assert_eq!(snap.reasons, vec![ReasonCode::Fatal]); - assert_eq!( - snap.fatal.as_ref().map(|f| f.code), - Some(FatalCode::InitializationFailed) - ); - - m.apply(HealthEvent::InitializationCompleted); - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - m.apply(HealthEvent::TickSucceeded); - let after = m.snapshot(); - assert!(!after.ready); - assert_eq!(after.phase, HealthPhase::Fatal); - assert_eq!( - after.fatal.as_ref().map(|f| f.code), - Some(FatalCode::InitializationFailed) - ); - } - - #[test] - fn checkpoint_reject_is_sticky_fatal_from_loading() { - let (mut m, _) = machine(); - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::InitializationCompleted); - m.apply(HealthEvent::CheckpointRejected { - detail: "digest mismatch".into(), - }); - assert_eq!(m.snapshot().phase, HealthPhase::Fatal); - - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - assert!(!m.snapshot().ready); - assert_eq!(m.snapshot().phase, HealthPhase::Fatal); - } - - #[test] - fn fatal_from_running_and_degraded_never_returns_to_ready() { - let (mut running, _) = machine(); - bring_ready(&mut running); - running.apply(HealthEvent::Fatal { - code: FatalCode::Unspecified, - detail: "network step invariant broken".into(), - }); - assert_eq!(running.snapshot().phase, HealthPhase::Fatal); - running.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - assert!(!running.snapshot().ready); - - let (mut degraded, clock) = machine(); - bring_ready(&mut degraded); - clock.advance(Duration::from_millis(100)); - assert_eq!(degraded.snapshot().phase, HealthPhase::Degraded); - degraded.apply(HealthEvent::Fatal { - code: FatalCode::Unspecified, - detail: "boom".into(), - }); - degraded.apply(HealthEvent::IngressObserved); - degraded.apply(HealthEvent::BeginDrain); - let snap = degraded.snapshot(); - assert_eq!(snap.phase, HealthPhase::Fatal); - assert!(!snap.ready); - assert!(snap.live); - } - - #[test] - fn fatal_from_draining_stays_fatal() { - let (mut m, _) = machine(); - bring_ready(&mut m); - m.apply(HealthEvent::BeginDrain); - m.apply(HealthEvent::Fatal { - code: FatalCode::Unspecified, - detail: "flush failed".into(), - }); - let snap = m.snapshot(); - assert_eq!(snap.phase, HealthPhase::Fatal); - assert!(!snap.ready); - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); - assert_eq!(m.snapshot().phase, HealthPhase::Fatal); - } - - #[test] - fn second_fatal_does_not_replace_the_first() { - let (mut m, _) = machine(); - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::InitializationFailed { - detail: "first".into(), - }); - m.apply(HealthEvent::Fatal { - code: FatalCode::Unspecified, - detail: "second".into(), - }); - let fatal = m.snapshot().fatal.expect("fatal"); - assert_eq!(fatal.code, FatalCode::InitializationFailed); - assert_eq!(fatal.detail, "first"); - } - - #[test] - fn prometheus_labels_are_low_cardinality_and_omit_detail() { - let (mut m, _) = machine(); - m.apply(HealthEvent::ProcessStarted); - m.apply(HealthEvent::InitializationFailed { - detail: "secret token xyzzy should never be a label".into(), - }); - let text = m.snapshot().prometheus_text(); - assert!(text.contains("brainstem_live 1")); - assert!(text.contains("brainstem_ready 0")); - assert!(text.contains("brainstem_fatal 1")); - assert!(text.contains("phase=\"fatal\"")); - assert!(!text.contains("xyzzy")); - assert!(!text.contains("secret token")); - assert!(text.contains("reason=\"stale_input\"")); - assert!(text.contains("reason=\"overload\"")); - } - - #[test] - fn try_snapshot_does_not_block_on_write_lock() { - let handle = HealthHandle::started(limits()); - let start = Instant::now(); - { - let _guard = handle.lock_write_for_test(); - assert!(handle.try_snapshot().is_none()); - } - assert!(start.elapsed() < Duration::from_millis(50)); - assert!(handle.try_snapshot().is_some()); - let snap = handle.snapshot(); - assert!(snap.live); - assert!(!snap.ready); - } - - #[test] - fn example_snapshots_match_documented_shapes() { - let (mut healthy, _) = machine(); - bring_ready(&mut healthy); - let healthy = healthy.snapshot(); - assert_eq!( - serde_json::to_value(&healthy).unwrap(), - serde_json::json!({ - "live": true, - "ready": true, - "phase": "running", - "reasons": [], - "last_successful_tick_ms": 0, - "checkpoint": { "id": "soma16", "digest": "abc123" }, - "input_freshness": { "age_ms": 0, "stale": false }, - "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, - "fatal": null, - "observed_at_ms": 0 - }) - ); - - let (mut degraded, clock) = machine(); - bring_ready(&mut degraded); - degraded.apply(HealthEvent::QueuePressure { - depth: 95, - capacity: 100, - }); - clock.advance(Duration::from_millis(100)); - let degraded = degraded.snapshot(); - assert_eq!(degraded.phase, HealthPhase::Degraded); - assert_eq!( - degraded.reasons, - vec![ReasonCode::StaleInput, ReasonCode::Overload] - ); - assert!(degraded.ready); - - let (mut fatal, _) = machine(); - fatal.apply(HealthEvent::ProcessStarted); - fatal.apply(HealthEvent::InitializationCompleted); - fatal.apply(HealthEvent::CheckpointRejected { - detail: "blank weights".into(), - }); - let fatal = fatal.snapshot(); - assert_eq!(fatal.phase, HealthPhase::Fatal); - assert!(!fatal.ready); - assert!(fatal.live); - assert_eq!(fatal.fatal.unwrap().code, FatalCode::CheckpointInvalid); - } -} diff --git a/src/health/mod.rs b/src/health/mod.rs new file mode 100644 index 0000000..b015c55 --- /dev/null +++ b/src/health/mod.rs @@ -0,0 +1,614 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +//! Process health: liveness, readiness, recoverable degradation, and sticky fatal state. +//! +//! Supervisors should treat [`HealthSnapshot::live`] and [`HealthSnapshot::ready`] as +//! independent. A live process has not necessarily loaded a valid checkpoint. Recoverable +//! reasons (`stale_input`, `overload`) clear only after the condition is observed healthy. +//! Fatal and draining states never return to ready in the same process. +//! +//! Snapshot reads take a separate lock from the tick loop's backend and network, so they +//! do not wait on `StimulusSource` or `SpikingNetwork::step`. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, RwLock}; +use std::time::{Duration, Instant}; + +use serde::{Deserialize, Serialize}; + +/// Monotonic clock used by the health state machine. +pub trait Clock: Send + Sync { + fn now(&self) -> Instant; +} + +/// Wall-clock monotonic clock. +#[derive(Debug, Clone, Copy, Default)] +pub struct SystemClock; + +impl Clock for SystemClock { + fn now(&self) -> Instant { + Instant::now() + } +} + +/// Test clock. Clone and share the same offset with a [`HealthMachine`]. +#[derive(Debug, Clone)] +pub struct FakeClock { + origin: Instant, + offset_nanos: Arc, +} + +impl FakeClock { + pub fn new() -> Self { + Self::default() + } + + pub fn advance(&self, duration: Duration) { + let add = u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX); + self.offset_nanos.fetch_add(add, Ordering::SeqCst); + } +} + +impl Default for FakeClock { + fn default() -> Self { + Self { + origin: Instant::now(), + offset_nanos: Arc::new(AtomicU64::new(0)), + } + } +} + +impl Clock for FakeClock { + fn now(&self) -> Instant { + self.origin + Duration::from_nanos(self.offset_nanos.load(Ordering::SeqCst)) + } +} + +/// Thresholds for recoverable degradation. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct HealthLimits { + /// Ingress older than this marks `stale_input`. + pub stale_after: Duration, + /// Queue fill ratio that *enters* overload (`depth / capacity`). + pub overload_high: f64, + /// Queue fill ratio that *exits* overload (hysteresis; must be `<= overload_high`). + pub overload_low: f64, +} + +impl Default for HealthLimits { + fn default() -> Self { + Self { + stale_after: Duration::from_millis(500), + overload_high: 0.90, + overload_low: 0.70, + } + } +} + +impl HealthLimits { + /// Replace non-finite or inverted watermarks with the built-in defaults. + pub fn sanitized(self) -> Self { + let high = finite_or(self.overload_high, 0.90); + let low = finite_or(self.overload_low, 0.70); + if high >= low { + Self { + stale_after: self.stale_after, + overload_high: high, + overload_low: low, + } + } else { + Self::default() + } + } +} + +fn finite_or(value: f64, fallback: f64) -> f64 { + if value.is_finite() { value } else { fallback } +} + +/// Coarse phase derived from the snapshot. Stable, low-cardinality. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum HealthPhase { + Starting, + LoadingCheckpoint, + Running, + Degraded, + Draining, + Fatal, +} + +impl HealthPhase { + pub const ALL: [HealthPhase; 6] = [ + HealthPhase::Starting, + HealthPhase::LoadingCheckpoint, + HealthPhase::Running, + HealthPhase::Degraded, + HealthPhase::Draining, + HealthPhase::Fatal, + ]; + + pub fn as_str(self) -> &'static str { + match self { + HealthPhase::Starting => "starting", + HealthPhase::LoadingCheckpoint => "loading_checkpoint", + HealthPhase::Running => "running", + HealthPhase::Degraded => "degraded", + HealthPhase::Draining => "draining", + HealthPhase::Fatal => "fatal", + } + } +} + +/// Stable reason codes. Never put detailed error text in metric labels — use these codes. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ReasonCode { + Starting, + CheckpointPending, + StaleInput, + Overload, + Draining, + Fatal, +} + +impl ReasonCode { + pub const RECOVERABLE: [ReasonCode; 2] = [ReasonCode::StaleInput, ReasonCode::Overload]; + + pub fn as_str(self) -> &'static str { + match self { + ReasonCode::Starting => "starting", + ReasonCode::CheckpointPending => "checkpoint_pending", + ReasonCode::StaleInput => "stale_input", + ReasonCode::Overload => "overload", + ReasonCode::Draining => "draining", + ReasonCode::Fatal => "fatal", + } + } +} + +/// Sticky fatal class. Low-cardinality; details live on [`FatalState::detail`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum FatalCode { + InitializationFailed, + CheckpointInvalid, + Unspecified, +} + +impl FatalCode { + pub fn as_str(self) -> &'static str { + match self { + FatalCode::InitializationFailed => "initialization_failed", + FatalCode::CheckpointInvalid => "checkpoint_invalid", + FatalCode::Unspecified => "unspecified", + } + } +} + +/// Identity of the loaded checkpoint. Digest may be absent until real validation lands. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct CheckpointIdentity { + pub id: String, + pub digest: Option, +} + +/// Fatal snapshot payload. `detail` is for logs/JSON, never a Prometheus label. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct FatalState { + pub code: FatalCode, + pub detail: String, +} + +/// Ingress freshness relative to the health clock. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct InputFreshness { + pub age_ms: Option, + pub stale: bool, +} + +/// Bounded-queue pressure. `capacity == 0` means "no queue instrumented". +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct QueuePressure { + pub depth: u64, + pub capacity: u64, + pub ratio: Option, + pub overloaded: bool, +} + +/// Machine-readable health view for supervisors and the control surface. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct HealthSnapshot { + pub live: bool, + pub ready: bool, + pub phase: HealthPhase, + pub reasons: Vec, + pub last_successful_tick_ms: Option, + /// Milliseconds since the last successful tick (`None` if none yet). + pub tick_age_ms: Option, + pub checkpoint: Option, + pub input_freshness: InputFreshness, + pub queue_pressure: QueuePressure, + pub fatal: Option, + pub observed_at_ms: u64, +} + +impl HealthSnapshot { + /// Prometheus text exposition. Labels are stable reason/phase codes only. + pub fn prometheus_text(&self) -> String { + let mut out = String::default(); + push_gauge( + &mut out, + "brainstem_live", + "1 if the process health reporter is running.", + u8::from(self.live), + ); + push_gauge( + &mut out, + "brainstem_ready", + "1 if initialized, checkpoint-valid, not draining, not fatal.", + u8::from(self.ready), + ); + + out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); + out.push_str("# TYPE brainstem_phase gauge\n"); + for phase in HealthPhase::ALL { + out.push_str(&format!( + "brainstem_phase{{phase=\"{}\"}} {}\n", + phase.as_str(), + u8::from(self.phase == phase) + )); + } + + out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); + out.push_str("# TYPE brainstem_degraded gauge\n"); + for reason in ReasonCode::RECOVERABLE { + let active = self.reasons.contains(&reason); + out.push_str(&format!( + "brainstem_degraded{{reason=\"{}\"}} {}\n", + reason.as_str(), + u8::from(active) + )); + } + + push_gauge( + &mut out, + "brainstem_fatal", + "1 if this process has entered a sticky fatal state.", + u8::from(self.fatal.is_some()), + ); + push_gauge( + &mut out, + "brainstem_last_successful_tick_ms", + "Milliseconds from process start until the last successful tick (not tick age).", + self.last_successful_tick_ms.unwrap_or(0), + ); + push_gauge( + &mut out, + "brainstem_tick_age_ms", + "Milliseconds since the last successful tick; 0 if none yet.", + self.tick_age_ms.unwrap_or(0), + ); + push_gauge( + &mut out, + "brainstem_input_age_ms", + "Age of last ingress (or checkpoint, if none) in milliseconds.", + self.input_freshness.age_ms.unwrap_or(0), + ); + push_gauge( + &mut out, + "brainstem_queue_depth", + "Ingress queue depth.", + self.queue_pressure.depth, + ); + push_gauge( + &mut out, + "brainstem_queue_capacity", + "Ingress queue capacity.", + self.queue_pressure.capacity, + ); + out + } +} + +fn push_gauge(out: &mut String, name: &str, help: &str, value: impl std::fmt::Display) { + out.push_str("# HELP "); + out.push_str(name); + out.push(' '); + out.push_str(help); + out.push('\n'); + out.push_str("# TYPE "); + out.push_str(name); + out.push_str(" gauge\n"); + out.push_str(name); + out.push(' '); + out.push_str(&value.to_string()); + out.push('\n'); +} + +/// State-machine events. Tick-loop I/O never runs while these are applied. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HealthEvent { + ProcessStarted, + InitializationCompleted, + InitializationFailed { detail: String }, + CheckpointValidated { identity: CheckpointIdentity }, + CheckpointRejected { detail: String }, + TickSucceeded, + IngressObserved, + QueuePressure { depth: u64, capacity: u64 }, + BeginDrain, + Fatal { code: FatalCode, detail: String }, +} + +/// Pure health state machine. Drive it with [`FakeClock`] in tests. +pub struct HealthMachine { + clock: Arc, + limits: HealthLimits, + started: bool, + initialized: bool, + checkpoint_ok: bool, + draining: bool, + started_at: Option, + checkpoint_at: Option, + last_tick: Option, + last_ingress: Option, + checkpoint: Option, + queue_depth: u64, + queue_capacity: u64, + overloaded: bool, + fatal: Option, +} + +impl HealthMachine { + pub fn new(clock: impl Clock + 'static, limits: HealthLimits) -> Self { + Self { + clock: Arc::new(clock), + limits: limits.sanitized(), + started: false, + initialized: false, + checkpoint_ok: false, + draining: false, + started_at: None, + checkpoint_at: None, + last_tick: None, + last_ingress: None, + checkpoint: None, + queue_depth: 0, + queue_capacity: 0, + overloaded: false, + fatal: None, + } + } + + pub fn apply(&mut self, event: HealthEvent) { + let now = self.clock.now(); + match event { + HealthEvent::TickSucceeded + | HealthEvent::IngressObserved + | HealthEvent::QueuePressure { .. } => self.apply_runtime(event, now), + other => self.apply_lifecycle(other, now), + } + } + + fn apply_lifecycle(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::ProcessStarted => { + if !self.started { + self.started = true; + self.started_at = Some(now); + } + } + HealthEvent::InitializationCompleted => { + if self.fatal.is_none() && !self.draining { + self.initialized = true; + } + } + HealthEvent::InitializationFailed { detail } => { + self.enter_fatal(FatalCode::InitializationFailed, detail); + } + HealthEvent::CheckpointValidated { identity } => { + if self.fatal.is_none() && !self.draining && self.initialized { + self.checkpoint = Some(identity); + self.checkpoint_ok = true; + self.checkpoint_at = Some(now); + } + } + HealthEvent::CheckpointRejected { detail } => { + self.enter_fatal(FatalCode::CheckpointInvalid, detail); + } + HealthEvent::BeginDrain => { + if self.fatal.is_none() { + self.draining = true; + } + } + HealthEvent::Fatal { code, detail } => { + self.enter_fatal(code, detail); + } + HealthEvent::TickSucceeded + | HealthEvent::IngressObserved + | HealthEvent::QueuePressure { .. } => {} + } + } + + fn apply_runtime(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::TickSucceeded => self.last_tick = Some(now), + HealthEvent::IngressObserved => self.last_ingress = Some(now), + HealthEvent::QueuePressure { depth, capacity } => { + self.queue_depth = depth; + self.queue_capacity = capacity; + self.overloaded = next_overload( + self.overloaded, + depth, + capacity, + self.limits.overload_high, + self.limits.overload_low, + ); + } + _ => {} + } + } + + pub fn snapshot(&self) -> HealthSnapshot { + let now = self.clock.now(); + let origin = self.started_at.unwrap_or(now); + let ms = |t: Instant| duration_ms(t.saturating_duration_since(origin)); + let age_base = self.last_ingress.or(self.checkpoint_at); + let age_ms = age_base.map(|t| duration_ms(now.saturating_duration_since(t))); + let stale = self.compute_stale(now); + let live = self.started; + let ready = live + && self.initialized + && self.checkpoint_ok + && !self.draining + && self.fatal.is_none(); + let ratio = if self.queue_capacity == 0 { + None + } else { + Some(self.queue_depth as f64 / self.queue_capacity as f64) + }; + + HealthSnapshot { + live, + ready, + phase: self.phase(stale), + reasons: self.reasons(stale), + last_successful_tick_ms: self.last_tick.map(ms), + tick_age_ms: self + .last_tick + .map(|t| duration_ms(now.saturating_duration_since(t))), + checkpoint: self.checkpoint.clone(), + input_freshness: InputFreshness { age_ms, stale }, + queue_pressure: QueuePressure { + depth: self.queue_depth, + capacity: self.queue_capacity, + ratio, + overloaded: self.overloaded, + }, + fatal: self.fatal.clone(), + observed_at_ms: ms(now), + } + } + + fn reasons(&self, stale: bool) -> Vec { + if self.fatal.is_some() { + return vec![ReasonCode::Fatal]; + } + let mut reasons = Vec::default(); + if !self.initialized { + reasons.push(ReasonCode::Starting); + } else if !self.checkpoint_ok { + reasons.push(ReasonCode::CheckpointPending); + } + if self.draining { + reasons.push(ReasonCode::Draining); + } + if stale { + reasons.push(ReasonCode::StaleInput); + } + if self.overloaded { + reasons.push(ReasonCode::Overload); + } + reasons + } + + fn phase(&self, stale: bool) -> HealthPhase { + if self.fatal.is_some() { + HealthPhase::Fatal + } else if self.draining { + HealthPhase::Draining + } else if !self.initialized { + HealthPhase::Starting + } else if !self.checkpoint_ok { + HealthPhase::LoadingCheckpoint + } else if stale || self.overloaded { + HealthPhase::Degraded + } else { + HealthPhase::Running + } + } + + fn enter_fatal(&mut self, code: FatalCode, detail: String) { + if self.fatal.is_some() { + return; + } + self.fatal = Some(FatalState { code, detail }); + self.checkpoint_ok = false; + } + + fn compute_stale(&self, now: Instant) -> bool { + if !self.initialized || !self.checkpoint_ok { + return false; + } + let baseline = match self.last_ingress.or(self.checkpoint_at) { + Some(t) => t, + None => return false, + }; + now.saturating_duration_since(baseline) >= self.limits.stale_after + } +} + +fn next_overload(currently: bool, depth: u64, capacity: u64, high: f64, low: f64) -> bool { + if capacity == 0 { + return false; + } + let ratio = depth as f64 / capacity as f64; + if currently { + ratio > low + } else { + ratio >= high + } +} + +fn duration_ms(d: Duration) -> u64 { + u64::try_from(d.as_millis()).unwrap_or(u64::MAX) +} + +/// Cloneable, non-tick-blocking handle for supervisors and the control surface. +#[derive(Clone)] +pub struct HealthHandle { + inner: Arc>, +} + +impl HealthHandle { + /// Live process, not yet ready. Used when the daemon is constructed. + pub fn started(limits: HealthLimits) -> Self { + let mut machine = HealthMachine::new(SystemClock, limits); + machine.apply(HealthEvent::ProcessStarted); + Self { + inner: Arc::new(RwLock::new(machine)), + } + } + + pub fn from_machine(machine: HealthMachine) -> Self { + Self { + inner: Arc::new(RwLock::new(machine)), + } + } + + pub fn apply(&self, event: HealthEvent) { + let mut guard = self.inner.write().unwrap_or_else(|e| e.into_inner()); + guard.apply(event); + } + + /// Clone the current snapshot. May wait only for an in-flight `apply` (no I/O). + pub fn snapshot(&self) -> HealthSnapshot { + let guard = self.inner.read().unwrap_or_else(|e| e.into_inner()); + guard.snapshot() + } + + /// Never waits. Returns `None` if a writer currently holds the lock. + pub fn try_snapshot(&self) -> Option { + self.inner.try_read().ok().map(|guard| guard.snapshot()) + } + + #[cfg(test)] + pub(crate) fn lock_write_for_test(&self) -> std::sync::RwLockWriteGuard<'_, HealthMachine> { + self.inner.write().unwrap_or_else(|e| e.into_inner()) + } +} + +#[cfg(test)] +mod tests; diff --git a/src/health/tests.rs b/src/health/tests.rs new file mode 100644 index 0000000..ead29e6 --- /dev/null +++ b/src/health/tests.rs @@ -0,0 +1,450 @@ + +use super::*; +use std::time::Duration; + +fn limits() -> HealthLimits { + HealthLimits { + stale_after: Duration::from_millis(100), + overload_high: 0.90, + overload_low: 0.70, + } +} + +fn machine() -> (HealthMachine, FakeClock) { + let clock = FakeClock::default(); + let machine = HealthMachine::new(clock.clone(), limits()); + (machine, clock) +} + +fn ckpt() -> CheckpointIdentity { + CheckpointIdentity { + id: "soma16".into(), + digest: Some("abc123".into()), + } +} + +fn bring_ready(m: &mut HealthMachine) { + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::IngressObserved); + m.apply(HealthEvent::TickSucceeded); +} + +#[test] +fn process_start_is_live_but_not_ready() { + let (mut m, _) = machine(); + let before = m.snapshot(); + assert!(!before.live); + assert!(!before.ready); + assert_eq!(before.phase, HealthPhase::Starting); + + m.apply(HealthEvent::ProcessStarted); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Starting); + assert_eq!(snap.reasons, vec![ReasonCode::Starting]); +} + +#[test] +fn initialization_does_not_imply_readiness() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::LoadingCheckpoint); + assert_eq!(snap.reasons, vec![ReasonCode::CheckpointPending]); +} + +#[test] +fn checkpoint_validation_makes_ready() { + let (mut m, _) = machine(); + bring_ready(&mut m); + let snap = m.snapshot(); + assert!(snap.live); + assert!(snap.ready); + assert_eq!(snap.phase, HealthPhase::Running); + assert!(snap.reasons.is_empty()); + assert_eq!( + snap.checkpoint.as_ref().map(|c| c.id.as_str()), + Some("soma16") + ); + assert_eq!(snap.last_successful_tick_ms, Some(0)); + assert!(!snap.input_freshness.stale); +} + +#[test] +fn checkpoint_before_init_is_ignored() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + let snap = m.snapshot(); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Starting); + assert!(snap.checkpoint.is_none()); +} + +#[test] +fn stale_input_degrades_and_recovers_only_after_fresh_ingress() { + let (mut m, clock) = machine(); + bring_ready(&mut m); + + clock.advance(Duration::from_millis(99)); + let still_ok = m.snapshot(); + assert_eq!(still_ok.phase, HealthPhase::Running); + assert!(still_ok.ready); + assert!(!still_ok.input_freshness.stale); + + clock.advance(Duration::from_millis(1)); + let degraded = m.snapshot(); + assert!(degraded.live); + assert!( + degraded.ready, + "stale input is recoverable degradation, not unreadiness" + ); + assert_eq!(degraded.phase, HealthPhase::Degraded); + assert_eq!(degraded.reasons, vec![ReasonCode::StaleInput]); + assert!(degraded.input_freshness.stale); + + clock.advance(Duration::from_millis(50)); + m.apply(HealthEvent::TickSucceeded); + let still_stale = m.snapshot(); + assert!( + still_stale.input_freshness.stale, + "ticks without ingress must not clear stale_input" + ); + + m.apply(HealthEvent::IngressObserved); + let recovered = m.snapshot(); + assert_eq!(recovered.phase, HealthPhase::Running); + assert!(!recovered.input_freshness.stale); + assert!(recovered.reasons.is_empty()); + assert!(recovered.ready); +} + +#[test] +fn overload_uses_hysteresis_and_clears_only_below_low_watermark() { + let (mut m, _) = machine(); + bring_ready(&mut m); + + m.apply(HealthEvent::QueuePressure { + depth: 89, + capacity: 100, + }); + assert_eq!(m.snapshot().phase, HealthPhase::Running); + + m.apply(HealthEvent::QueuePressure { + depth: 90, + capacity: 100, + }); + let high = m.snapshot(); + assert_eq!(high.phase, HealthPhase::Degraded); + assert!(high.ready); + assert_eq!(high.reasons, vec![ReasonCode::Overload]); + assert!(high.queue_pressure.overloaded); + + m.apply(HealthEvent::QueuePressure { + depth: 80, + capacity: 100, + }); + let mid = m.snapshot(); + assert!( + mid.queue_pressure.overloaded, + "must stay overloaded between high and low watermarks" + ); + assert_eq!(mid.phase, HealthPhase::Degraded); + + m.apply(HealthEvent::QueuePressure { + depth: 70, + capacity: 100, + }); + let recovered = m.snapshot(); + assert!(!recovered.queue_pressure.overloaded); + assert_eq!(recovered.phase, HealthPhase::Running); + assert!(recovered.ready); +} + +#[test] +fn zero_capacity_queue_is_not_overload() { + let (mut m, _) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::QueuePressure { + depth: 0, + capacity: 0, + }); + let snap = m.snapshot(); + assert!(!snap.queue_pressure.overloaded); + assert_eq!(snap.queue_pressure.ratio, None); + assert_eq!(snap.phase, HealthPhase::Running); +} + +#[test] +fn combined_degradation_clears_independently() { + let (mut m, clock) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::QueuePressure { + depth: 95, + capacity: 100, + }); + clock.advance(Duration::from_millis(100)); + let both = m.snapshot(); + assert_eq!( + both.reasons, + vec![ReasonCode::StaleInput, ReasonCode::Overload] + ); + + m.apply(HealthEvent::IngressObserved); + let only_overload = m.snapshot(); + assert_eq!(only_overload.reasons, vec![ReasonCode::Overload]); + assert_eq!(only_overload.phase, HealthPhase::Degraded); + + m.apply(HealthEvent::QueuePressure { + depth: 10, + capacity: 100, + }); + let clear = m.snapshot(); + assert!(clear.reasons.is_empty()); + assert_eq!(clear.phase, HealthPhase::Running); +} + +#[test] +fn draining_drops_readiness_and_does_not_return_to_ready() { + let (mut m, _) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::BeginDrain); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Draining); + assert!(snap.reasons.contains(&ReasonCode::Draining)); + + m.apply(HealthEvent::TickSucceeded); + m.apply(HealthEvent::IngressObserved); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::InitializationCompleted); + let after = m.snapshot(); + assert!(!after.ready); + assert_eq!(after.phase, HealthPhase::Draining); +} + +#[test] +fn initialization_failure_is_sticky_fatal() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationFailed { + detail: "socket bind exploded with secret=hunter2".into(), + }); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Fatal); + assert_eq!(snap.reasons, vec![ReasonCode::Fatal]); + assert_eq!( + snap.fatal.as_ref().map(|f| f.code), + Some(FatalCode::InitializationFailed) + ); + + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::TickSucceeded); + let after = m.snapshot(); + assert!(!after.ready); + assert_eq!(after.phase, HealthPhase::Fatal); + assert_eq!( + after.fatal.as_ref().map(|f| f.code), + Some(FatalCode::InitializationFailed) + ); +} + +#[test] +fn checkpoint_reject_is_sticky_fatal_from_loading() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointRejected { + detail: "digest mismatch".into(), + }); + assert_eq!(m.snapshot().phase, HealthPhase::Fatal); + + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + assert!(!m.snapshot().ready); + assert_eq!(m.snapshot().phase, HealthPhase::Fatal); +} + +#[test] +fn fatal_from_running_and_degraded_never_returns_to_ready() { + let (mut running, _) = machine(); + bring_ready(&mut running); + running.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "network step invariant broken".into(), + }); + assert_eq!(running.snapshot().phase, HealthPhase::Fatal); + running.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + assert!(!running.snapshot().ready); + + let (mut degraded, clock) = machine(); + bring_ready(&mut degraded); + clock.advance(Duration::from_millis(100)); + assert_eq!(degraded.snapshot().phase, HealthPhase::Degraded); + degraded.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "boom".into(), + }); + degraded.apply(HealthEvent::IngressObserved); + degraded.apply(HealthEvent::BeginDrain); + let snap = degraded.snapshot(); + assert_eq!(snap.phase, HealthPhase::Fatal); + assert!(!snap.ready); + assert!(snap.live); +} + +#[test] +fn fatal_from_draining_stays_fatal() { + let (mut m, _) = machine(); + bring_ready(&mut m); + m.apply(HealthEvent::BeginDrain); + m.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "flush failed".into(), + }); + let snap = m.snapshot(); + assert_eq!(snap.phase, HealthPhase::Fatal); + assert!(!snap.ready); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + assert_eq!(m.snapshot().phase, HealthPhase::Fatal); +} + +#[test] +fn second_fatal_does_not_replace_the_first() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationFailed { + detail: "first".into(), + }); + m.apply(HealthEvent::Fatal { + code: FatalCode::Unspecified, + detail: "second".into(), + }); + let fatal = m.snapshot().fatal.expect("fatal"); + assert_eq!(fatal.code, FatalCode::InitializationFailed); + assert_eq!(fatal.detail, "first"); +} + +#[test] +fn prometheus_labels_are_low_cardinality_and_omit_detail() { + let (mut m, _) = machine(); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationFailed { + detail: "secret token xyzzy should never be a label".into(), + }); + let text = m.snapshot().prometheus_text(); + assert!(text.contains("brainstem_live 1")); + assert!(text.contains("brainstem_ready 0")); + assert!(text.contains("brainstem_fatal 1")); + assert!(text.contains("phase=\"fatal\"")); + assert!(!text.contains("xyzzy")); + assert!(!text.contains("secret token")); + assert!(text.contains("brainstem_tick_age_ms")); + assert!(text.contains("reason=\"stale_input\"")); + assert!(text.contains("reason=\"overload\"")); +} + +#[test] +fn try_snapshot_does_not_block_on_write_lock() { + let handle = HealthHandle::started(limits()); + let start = Instant::now(); + { + let _guard = handle.lock_write_for_test(); + assert!(handle.try_snapshot().is_none()); + } + assert!(start.elapsed() < Duration::from_millis(50)); + assert!(handle.try_snapshot().is_some()); + let snap = handle.snapshot(); + assert!(snap.live); + assert!(!snap.ready); +} + +#[test] +fn example_snapshots_match_documented_shapes() { + let (mut healthy, _) = machine(); + bring_ready(&mut healthy); + let healthy = healthy.snapshot(); + assert_eq!( + serde_json::to_value(&healthy).unwrap(), + serde_json::json!({ + "live": true, + "ready": true, + "phase": "running", + "reasons": [], + "last_successful_tick_ms": 0, + "tick_age_ms": 0, + "checkpoint": { "id": "soma16", "digest": "abc123" }, + "input_freshness": { "age_ms": 0, "stale": false }, + "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, + "fatal": null, + "observed_at_ms": 0 + }) + ); + + let (mut degraded, clock) = machine(); + bring_ready(&mut degraded); + degraded.apply(HealthEvent::QueuePressure { + depth: 95, + capacity: 100, + }); + clock.advance(Duration::from_millis(100)); + let degraded = degraded.snapshot(); + assert_eq!(degraded.phase, HealthPhase::Degraded); + assert_eq!( + degraded.reasons, + vec![ReasonCode::StaleInput, ReasonCode::Overload] + ); + assert!(degraded.ready); + + let (mut fatal, _) = machine(); + fatal.apply(HealthEvent::ProcessStarted); + fatal.apply(HealthEvent::InitializationCompleted); + fatal.apply(HealthEvent::CheckpointRejected { + detail: "blank weights".into(), + }); + let fatal = fatal.snapshot(); + assert_eq!(fatal.phase, HealthPhase::Fatal); + assert!(!fatal.ready); + assert!(fatal.live); + assert_eq!(fatal.fatal.unwrap().code, FatalCode::CheckpointInvalid); +} + +#[test] +fn missing_digest_serializes_as_json_null() { + let identity = CheckpointIdentity { + id: "soma16".into(), + digest: None, + }; + assert_eq!( + serde_json::to_value(&identity).unwrap(), + serde_json::json!({ "id": "soma16", "digest": null }) + ); +} + +#[test] +fn non_finite_overload_limits_are_sanitized() { + let clock = FakeClock::default(); + let mut machine = HealthMachine::new( + clock, + HealthLimits { + stale_after: Duration::from_millis(100), + overload_high: f64::NAN, + overload_low: f64::NAN, + }, + ); + bring_ready(&mut machine); + machine.apply(HealthEvent::QueuePressure { + depth: 95, + capacity: 100, + }); + assert!(machine.snapshot().queue_pressure.overloaded); +} From fc314aa6416aedf326badb989445acefd39d4349 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 05:56:14 +0000 Subject: [PATCH 04/18] fix: split health helpers to clear rustfmt and Codacy complexity Extract lifecycle, probe routing, and tick-loop helpers so functions stay under Codacy's cyclomatic threshold, and drop the leading blank line that failed `cargo fmt --check`. Co-authored-by: Raul Cardenas Montoya --- src/control.rs | 67 ++++++++++------- src/daemon.rs | 179 +++++++++++++++++++++++++------------------- src/health/mod.rs | 77 +++++++++++++------ src/health/tests.rs | 1 - 4 files changed, 197 insertions(+), 127 deletions(-) diff --git a/src/control.rs b/src/control.rs index ecdaafa..3ac78e2 100644 --- a/src/control.rs +++ b/src/control.rs @@ -56,22 +56,7 @@ pub async fn serve_listener( } } accepted = listener.accept() => { - match accepted { - Ok((stream, _)) => { - let Ok(permit) = slots.clone().try_acquire_owned() else { - drop(stream); - continue; - }; - let health = health.clone(); - tokio::spawn(async move { - let _permit = permit; - if let Err(e) = handle_connection(stream, &health).await { - warn!("control connection failed: {e}"); - } - }); - } - Err(e) => warn!("control accept failed: {e}"), - } + spawn_accepted(accepted, &slots, &health); } } } @@ -79,6 +64,31 @@ pub async fn serve_listener( Ok(()) } +fn spawn_accepted( + accepted: std::io::Result<(TcpStream, std::net::SocketAddr)>, + slots: &Arc, + health: &HealthHandle, +) { + match accepted { + Ok((stream, _)) => spawn_control_conn(stream, slots, health), + Err(e) => warn!("control accept failed: {e}"), + } +} + +fn spawn_control_conn(stream: TcpStream, slots: &Arc, health: &HealthHandle) { + let Ok(permit) = slots.clone().try_acquire_owned() else { + drop(stream); + return; + }; + let health = health.clone(); + tokio::spawn(async move { + let _permit = permit; + if let Err(e) = handle_connection(stream, &health).await { + warn!("control connection failed: {e}"); + } + }); +} + async fn handle_connection(mut stream: TcpStream, health: &HealthHandle) -> Result<()> { let mut buf = [0u8; 1024]; let n = tokio::time::timeout(IO_TIMEOUT, read_request_line(&mut stream, &mut buf)) @@ -160,24 +170,25 @@ enum ParseResult<'a> { } fn parse_get_path(request: &str) -> ParseResult<'_> { - let line = match request.lines().next() { - Some(line) => line, - None => return ParseResult::Invalid, + let Some(line) = request.lines().next() else { + return ParseResult::Invalid; }; + parse_request_line(line) +} + +fn parse_request_line(line: &str) -> ParseResult<'_> { let mut parts = line.split_whitespace(); - let method = match parts.next() { - Some(method) => method, - None => return ParseResult::Invalid, - }; - let target = match parts.next() { - Some(target) => target, - None => return ParseResult::Invalid, + let (Some(method), Some(target)) = (parts.next(), parts.next()) else { + return ParseResult::Invalid; }; if !method.eq_ignore_ascii_case("GET") { return ParseResult::NotGet; } - let path = target.split('?').next().unwrap_or(target); - ParseResult::Get(path) + ParseResult::Get(path_without_query(target)) +} + +fn path_without_query(target: &str) -> &str { + target.split('?').next().unwrap_or(target) } fn http_response(status: u16, content_type: &str, body: &[u8]) -> Vec { diff --git a/src/daemon.rs b/src/daemon.rs index e2591a0..20a1ae6 100644 --- a/src/daemon.rs +++ b/src/daemon.rs @@ -153,9 +153,7 @@ impl BrainstemDaemon { let mut backend = self.backend; let health = self.health; - if cfg.tick_rate_hz == 0 || cfg.tick_rate_hz > 1_000_000 { - anyhow::bail!("tick_rate_hz must be in range 1..=1_000_000"); - } + validate_tick_rate(cfg.tick_rate_hz)?; let (control_stop, control_task) = start_control(cfg.control_bind.as_deref(), health.clone()).await?; @@ -165,49 +163,9 @@ impl BrainstemDaemon { return Err(e); } - let tick_duration = Duration::from_nanos(1_000_000_000 / u64::from(cfg.tick_rate_hz)); - let mut ticker = time::interval(tick_duration); - ticker.set_missed_tick_behavior(time::MissedTickBehavior::Skip); - - let mut network = - SpikingNetwork::with_dimensions(cfg.lif_count, cfg.izh_count, cfg.channels); - let mut stimuli = vec![0.0; cfg.channels]; - let mut spike_buf: Vec = Vec::with_capacity(128); - - let mut shutdown = std::pin::pin!(shutdown_signal()); - - loop { - tokio::select! { - _ = ticker.tick() => { - run_tick( - &mut *backend.source, - &mut network, - &mut *backend.sink, - &mut stimuli, - &mut spike_buf, - &health, - ); - } - _ = &mut shutdown => { - info!("Termination signal received, shutting down"); - health.apply(HealthEvent::BeginDrain); - break; - } - } - } - + run_until_shutdown(&cfg, &mut backend, &health).await; stop_control(control_stop, control_task).await; - - // Explicit backend lifecycle hooks (flush sink, shutdown source) are invoked - // for custom backends. Current built-ins are no-ops, but this satisfies - // CodeAnt/CodeRabbit "missing cleanup" notes. - if let Err(e) = backend.sink.flush() { - warn!("Failed to flush spike sink on shutdown: {e}"); - } - if let Err(e) = backend.source.shutdown() { - warn!("Failed to shut down stimulus source: {e}"); - } - + shutdown_backend(&mut backend); Ok(()) } } @@ -335,6 +293,56 @@ fn packet_carries_input(packet: &IngressPacket) -> bool { .is_some_and(|mods| !mods.is_empty()) } +fn validate_tick_rate(tick_rate_hz: u32) -> Result<()> { + if tick_rate_hz == 0 || tick_rate_hz > 1_000_000 { + anyhow::bail!("tick_rate_hz must be in range 1..=1_000_000"); + } + Ok(()) +} + +async fn run_until_shutdown(cfg: &DaemonConfig, backend: &mut BackendPair, health: &HealthHandle) { + let tick_duration = Duration::from_nanos(1_000_000_000 / u64::from(cfg.tick_rate_hz)); + let mut ticker = time::interval(tick_duration); + ticker.set_missed_tick_behavior(time::MissedTickBehavior::Skip); + + let mut network = SpikingNetwork::with_dimensions(cfg.lif_count, cfg.izh_count, cfg.channels); + let mut stimuli = vec![0.0; cfg.channels]; + let mut spike_buf: Vec = Vec::with_capacity(128); + let mut shutdown = std::pin::pin!(shutdown_signal()); + + loop { + tokio::select! { + _ = ticker.tick() => { + run_tick( + &mut *backend.source, + &mut network, + &mut *backend.sink, + &mut stimuli, + &mut spike_buf, + health, + ); + } + _ = &mut shutdown => { + info!("Termination signal received, shutting down"); + health.apply(HealthEvent::BeginDrain); + break; + } + } + } +} + +fn shutdown_backend(backend: &mut BackendPair) { + // Explicit backend lifecycle hooks (flush sink, shutdown source) are invoked + // for custom backends. Current built-ins are no-ops, but this satisfies + // CodeAnt/CodeRabbit "missing cleanup" notes. + if let Err(e) = backend.sink.flush() { + warn!("Failed to flush spike sink on shutdown: {e}"); + } + if let Err(e) = backend.source.shutdown() { + warn!("Failed to shut down stimulus source: {e}"); + } +} + fn validate_neuron_count(config: &DaemonConfig) -> Result<()> { let total = config .lif_count @@ -369,60 +377,73 @@ fn run_tick( spike_buf: &mut Vec, health: &HealthHandle, ) { - let packet = match source.next_ingress() { + let Some(packet) = next_ingress_packet(source, health) else { + return; + }; + let modulators = decode_inputs(&packet, stimuli); + let Some(spike_ids) = step_network(network, stimuli, &modulators) else { + return; + }; + let now = SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default(); + fill_spike_buf(spike_buf, &spike_ids, now); + emit_tick(sink, spike_buf, &spike_ids, now, health); +} + +fn next_ingress_packet( + source: &mut dyn StimulusSource, + health: &HealthHandle, +) -> Option { + match source.next_ingress() { Ok(Some(p)) => { if packet_carries_input(&p) { health.apply(HealthEvent::IngressObserved); } - p + Some(p) } Ok(None) => { // Per StimulusSource contract: None means skip ingress this tick but still // advance the network with zeroed stimuli (maintains tick cadence). // decode_inputs will zero-fill the stimuli buffer based on the empty readout. - IngressPacket { + Some(IngressPacket { stimuli: Vec::new(), modulators: None, - } + }) } Err(e) => { warn!("Failed to receive from stimulus source: {e}"); - return; + None } - }; - - let modulators = decode_inputs(&packet, stimuli); - - // Note: decode_inputs already zero-fills any remaining channels when packet.stimuli is shorter. + } +} - let spike_ids = match network.step(stimuli, &modulators) { - Ok(spikes) => spikes, +fn step_network( + network: &mut SpikingNetwork, + stimuli: &[f32], + modulators: &NeuroModulators, +) -> Option> { + match network.step(stimuli, modulators) { + Ok(spikes) => Some(spikes), Err(e) => { error!("Network step failed: {e:?}"); - return; + None } - }; + } +} - // Single timestamp for both per-spike time and batch metadata (keeps them consistent). - let now = SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default(); +fn fill_spike_buf(spike_buf: &mut Vec, spike_ids: &[usize], now: Duration) { let tick = now.as_millis() as u64; - spike_buf.clear(); let mut dropped = 0usize; - for &idx in &spike_ids { + for &idx in spike_ids { match u16::try_from(idx) { - Ok(channel) => { - spike_buf.push(LocalSpikeEvent { - channel, - time: (tick & (u32::MAX as u64)) as u32, - strength: 1.0, - }); - } - Err(_) => { - dropped += 1; - } + Ok(channel) => spike_buf.push(LocalSpikeEvent { + channel, + time: (tick & (u32::MAX as u64)) as u32, + strength: 1.0, + }), + Err(_) => dropped += 1, } } if dropped > 0 { @@ -431,7 +452,15 @@ fn run_tick( dropped ); } +} +fn emit_tick( + sink: &mut dyn SpikeSink, + spike_buf: &[LocalSpikeEvent], + spike_ids: &[usize], + now: Duration, + health: &HealthHandle, +) { if spike_buf.is_empty() && !spike_ids.is_empty() { // Had spikes from network but all IDs were out of u16 range (dropped). // Nothing valid to publish; skip to avoid empty batch for dropped case. diff --git a/src/health/mod.rs b/src/health/mod.rs index b015c55..1b70c13 100644 --- a/src/health/mod.rs +++ b/src/health/mod.rs @@ -253,24 +253,10 @@ impl HealthSnapshot { out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); out.push_str("# TYPE brainstem_phase gauge\n"); - for phase in HealthPhase::ALL { - out.push_str(&format!( - "brainstem_phase{{phase=\"{}\"}} {}\n", - phase.as_str(), - u8::from(self.phase == phase) - )); - } - + write_phase_gauges(&mut out, self.phase); out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); out.push_str("# TYPE brainstem_degraded gauge\n"); - for reason in ReasonCode::RECOVERABLE { - let active = self.reasons.contains(&reason); - out.push_str(&format!( - "brainstem_degraded{{reason=\"{}\"}} {}\n", - reason.as_str(), - u8::from(active) - )); - } + write_degraded_gauges(&mut out, &self.reasons); push_gauge( &mut out, @@ -327,6 +313,26 @@ fn push_gauge(out: &mut String, name: &str, help: &str, value: impl std::fmt::Di out.push('\n'); } +fn write_phase_gauges(out: &mut String, current: HealthPhase) { + for phase in HealthPhase::ALL { + out.push_str(&format!( + "brainstem_phase{{phase=\"{}\"}} {}\n", + phase.as_str(), + u8::from(current == phase) + )); + } +} + +fn write_degraded_gauges(out: &mut String, reasons: &[ReasonCode]) { + for reason in ReasonCode::RECOVERABLE { + out.push_str(&format!( + "brainstem_degraded{{reason=\"{}\"}} {}\n", + reason.as_str(), + u8::from(reasons.contains(&reason)) + )); + } +} + /// State-machine events. Tick-loop I/O never runs while these are applied. #[derive(Debug, Clone, PartialEq, Eq)] pub enum HealthEvent { @@ -393,6 +399,19 @@ impl HealthMachine { } fn apply_lifecycle(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::ProcessStarted + | HealthEvent::InitializationCompleted + | HealthEvent::InitializationFailed { .. } => self.apply_boot(event, now), + HealthEvent::CheckpointValidated { .. } | HealthEvent::CheckpointRejected { .. } => { + self.apply_checkpoint(event, now) + } + HealthEvent::BeginDrain | HealthEvent::Fatal { .. } => self.apply_terminal(event), + _ => {} + } + } + + fn apply_boot(&mut self, event: HealthEvent, now: Instant) { match event { HealthEvent::ProcessStarted => { if !self.started { @@ -408,8 +427,14 @@ impl HealthMachine { HealthEvent::InitializationFailed { detail } => { self.enter_fatal(FatalCode::InitializationFailed, detail); } + _ => {} + } + } + + fn apply_checkpoint(&mut self, event: HealthEvent, now: Instant) { + match event { HealthEvent::CheckpointValidated { identity } => { - if self.fatal.is_none() && !self.draining && self.initialized { + if self.can_accept_checkpoint() { self.checkpoint = Some(identity); self.checkpoint_ok = true; self.checkpoint_at = Some(now); @@ -418,17 +443,23 @@ impl HealthMachine { HealthEvent::CheckpointRejected { detail } => { self.enter_fatal(FatalCode::CheckpointInvalid, detail); } + _ => {} + } + } + + fn can_accept_checkpoint(&self) -> bool { + self.fatal.is_none() && !self.draining && self.initialized + } + + fn apply_terminal(&mut self, event: HealthEvent) { + match event { HealthEvent::BeginDrain => { if self.fatal.is_none() { self.draining = true; } } - HealthEvent::Fatal { code, detail } => { - self.enter_fatal(code, detail); - } - HealthEvent::TickSucceeded - | HealthEvent::IngressObserved - | HealthEvent::QueuePressure { .. } => {} + HealthEvent::Fatal { code, detail } => self.enter_fatal(code, detail), + _ => {} } } diff --git a/src/health/tests.rs b/src/health/tests.rs index ead29e6..aa158a7 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -1,4 +1,3 @@ - use super::*; use std::time::Duration; From a9817866b3cf2a2ff047e1cc016993a60ced897d Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 05:59:20 +0000 Subject: [PATCH 05/18] fix: drop rejected control connections by returning from the helper DeepSource flagged an explicit `drop(stream)` after the complexity split. Returning from `spawn_control_conn` still closes the socket at the end of the function, which is the intended reject-when-full behavior. Co-authored-by: Raul Cardenas Montoya --- src/control.rs | 1 - 1 file changed, 1 deletion(-) diff --git a/src/control.rs b/src/control.rs index 3ac78e2..77ea6dd 100644 --- a/src/control.rs +++ b/src/control.rs @@ -77,7 +77,6 @@ fn spawn_accepted( fn spawn_control_conn(stream: TcpStream, slots: &Arc, health: &HealthHandle) { let Ok(permit) = slots.clone().try_acquire_owned() else { - drop(stream); return; }; let health = health.clone(); From 35774f77a05f46c501e4162779b9430bf0980393 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:05:44 +0000 Subject: [PATCH 06/18] fix: split remaining health and control helpers below Codacy CCN Break snapshot, phase/reason derivation, Prometheus text, and HTTP read/write into smaller functions so Codacy's medium complexity gate no longer fires on the health surface. Co-authored-by: Raul Cardenas Montoya --- src/control.rs | 66 ++++++++---- src/daemon.rs | 20 +++- src/health/mod.rs | 253 ++++++++++++++++++++++++++++------------------ 3 files changed, 220 insertions(+), 119 deletions(-) diff --git a/src/control.rs b/src/control.rs index 77ea6dd..736ff56 100644 --- a/src/control.rs +++ b/src/control.rs @@ -7,6 +7,7 @@ //! started from `BrainstemDaemon::run` when `control_bind` is set. Do not add //! a second server alongside this one. +use std::future::Future; use std::net::SocketAddr; use std::sync::Arc; use std::time::Duration; @@ -43,7 +44,7 @@ pub async fn serve_listener( .context("control listener has no local address")?; info!(%bound, "control surface listening (/livez /readyz /health /metrics)"); - if *shutdown.borrow() { + if shutdown_signaled(&shutdown) { return Ok(()); } @@ -51,7 +52,7 @@ pub async fn serve_listener( loop { tokio::select! { changed = shutdown.changed() => { - if changed.is_err() || *shutdown.borrow() { + if should_stop_control(changed, &shutdown) { break; } } @@ -64,6 +65,17 @@ pub async fn serve_listener( Ok(()) } +fn shutdown_signaled(shutdown: &watch::Receiver) -> bool { + *shutdown.borrow() +} + +fn should_stop_control( + changed: Result<(), watch::error::RecvError>, + shutdown: &watch::Receiver, +) -> bool { + changed.is_err() || shutdown_signaled(shutdown) +} + fn spawn_accepted( accepted: std::io::Result<(TcpStream, std::net::SocketAddr)>, slots: &Arc, @@ -89,40 +101,56 @@ fn spawn_control_conn(stream: TcpStream, slots: &Arc, health: &Health } async fn handle_connection(mut stream: TcpStream, health: &HealthHandle) -> Result<()> { + let req = read_http_request(&mut stream).await?; + let response = render_http(&req, &health.snapshot()); + write_http_response(&mut stream, &response).await +} + +async fn read_http_request(stream: &mut TcpStream) -> Result { let mut buf = [0u8; 1024]; - let n = tokio::time::timeout(IO_TIMEOUT, read_request_line(&mut stream, &mut buf)) - .await - .context("control read timed out")? - .context("control read failed")?; - let req = std::str::from_utf8(&buf[..n]).unwrap_or(""); - let snap = health.snapshot(); - let response = render_http(req, &snap); - tokio::time::timeout(IO_TIMEOUT, async { - stream.write_all(&response).await?; + let n = timed_io( + "control read timed out", + read_request_line(stream, &mut buf), + ) + .await?; + Ok(std::str::from_utf8(&buf[..n]).unwrap_or("").to_owned()) +} + +async fn write_http_response(stream: &mut TcpStream, response: &[u8]) -> Result<()> { + timed_io("control write timed out", async { + stream.write_all(response).await?; stream.flush().await?; - Ok::<_, std::io::Error>(()) + Ok(()) }) .await - .context("control write timed out")? - .context("control write failed")?; - Ok(()) +} + +async fn timed_io(timeout_msg: &'static str, fut: F) -> Result +where + F: Future>, +{ + tokio::time::timeout(IO_TIMEOUT, fut) + .await + .context(timeout_msg)? + .context("control I/O failed") } async fn read_request_line(stream: &mut TcpStream, buf: &mut [u8]) -> std::io::Result { let mut total = 0; while total < buf.len() { let n = stream.read(&mut buf[total..]).await?; - if n == 0 { - break; - } total += n; - if buf[..total].contains(&b'\n') { + if request_line_complete(n, &buf[..total]) { break; } } Ok(total) } +fn request_line_complete(read: usize, buf: &[u8]) -> bool { + read == 0 || buf.contains(&b'\n') +} + pub(crate) fn render_http(request: &str, snap: &HealthSnapshot) -> Vec { match parse_get_path(request) { ParseResult::Get(path) => { diff --git a/src/daemon.rs b/src/daemon.rs index 20a1ae6..7ad9ae6 100644 --- a/src/daemon.rs +++ b/src/daemon.rs @@ -225,19 +225,33 @@ async fn start_control( let Some(bind) = bind else { return Ok((None, None)); }; + let listener = bind_control(bind).await?; + Ok(spawn_control_task(listener, health)) +} + +async fn bind_control(bind: &str) -> Result { let addr: SocketAddr = bind .parse() .with_context(|| format!("invalid control_bind {bind}"))?; - let listener = tokio::net::TcpListener::bind(addr) + tokio::net::TcpListener::bind(addr) .await - .with_context(|| format!("failed to bind control surface on {addr}"))?; + .with_context(|| format!("failed to bind control surface on {addr}")) +} + +fn spawn_control_task( + listener: tokio::net::TcpListener, + health: HealthHandle, +) -> ( + Option>, + Option>, +) { let (tx, rx) = watch::channel(false); let task = tokio::spawn(async move { if let Err(e) = crate::control::serve_listener(listener, health, rx).await { warn!("control surface stopped: {e}"); } }); - Ok((Some(tx), Some(task))) + (Some(tx), Some(task)) } async fn stop_control( diff --git a/src/health/mod.rs b/src/health/mod.rs index 1b70c13..ce8161a 100644 --- a/src/health/mod.rs +++ b/src/health/mod.rs @@ -238,66 +238,76 @@ impl HealthSnapshot { /// Prometheus text exposition. Labels are stable reason/phase codes only. pub fn prometheus_text(&self) -> String { let mut out = String::default(); - push_gauge( - &mut out, - "brainstem_live", - "1 if the process health reporter is running.", - u8::from(self.live), - ); - push_gauge( - &mut out, - "brainstem_ready", - "1 if initialized, checkpoint-valid, not draining, not fatal.", - u8::from(self.ready), - ); - - out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); - out.push_str("# TYPE brainstem_phase gauge\n"); - write_phase_gauges(&mut out, self.phase); - out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); - out.push_str("# TYPE brainstem_degraded gauge\n"); - write_degraded_gauges(&mut out, &self.reasons); - - push_gauge( - &mut out, - "brainstem_fatal", - "1 if this process has entered a sticky fatal state.", - u8::from(self.fatal.is_some()), - ); - push_gauge( - &mut out, - "brainstem_last_successful_tick_ms", - "Milliseconds from process start until the last successful tick (not tick age).", - self.last_successful_tick_ms.unwrap_or(0), - ); - push_gauge( - &mut out, - "brainstem_tick_age_ms", - "Milliseconds since the last successful tick; 0 if none yet.", - self.tick_age_ms.unwrap_or(0), - ); - push_gauge( - &mut out, - "brainstem_input_age_ms", - "Age of last ingress (or checkpoint, if none) in milliseconds.", - self.input_freshness.age_ms.unwrap_or(0), - ); - push_gauge( - &mut out, - "brainstem_queue_depth", - "Ingress queue depth.", - self.queue_pressure.depth, - ); - push_gauge( - &mut out, - "brainstem_queue_capacity", - "Ingress queue capacity.", - self.queue_pressure.capacity, - ); + write_probe_gauges(&mut out, self); + write_labeled_gauges(&mut out, self); + write_runtime_gauges(&mut out, self); out } } +fn write_probe_gauges(out: &mut String, snap: &HealthSnapshot) { + push_gauge( + out, + "brainstem_live", + "1 if the process health reporter is running.", + u8::from(snap.live), + ); + push_gauge( + out, + "brainstem_ready", + "1 if initialized, checkpoint-valid, not draining, not fatal.", + u8::from(snap.ready), + ); +} + +fn write_labeled_gauges(out: &mut String, snap: &HealthSnapshot) { + out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); + out.push_str("# TYPE brainstem_phase gauge\n"); + write_phase_gauges(out, snap.phase); + out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); + out.push_str("# TYPE brainstem_degraded gauge\n"); + write_degraded_gauges(out, &snap.reasons); +} + +fn write_runtime_gauges(out: &mut String, snap: &HealthSnapshot) { + push_gauge( + out, + "brainstem_fatal", + "1 if this process has entered a sticky fatal state.", + u8::from(snap.fatal.is_some()), + ); + push_gauge( + out, + "brainstem_last_successful_tick_ms", + "Milliseconds from process start until the last successful tick (not tick age).", + snap.last_successful_tick_ms.unwrap_or(0), + ); + push_gauge( + out, + "brainstem_tick_age_ms", + "Milliseconds since the last successful tick; 0 if none yet.", + snap.tick_age_ms.unwrap_or(0), + ); + push_gauge( + out, + "brainstem_input_age_ms", + "Age of last ingress (or checkpoint, if none) in milliseconds.", + snap.input_freshness.age_ms.unwrap_or(0), + ); + push_gauge( + out, + "brainstem_queue_depth", + "Ingress queue depth.", + snap.queue_pressure.depth, + ); + push_gauge( + out, + "brainstem_queue_capacity", + "Ingress queue capacity.", + snap.queue_pressure.capacity, + ); +} + fn push_gauge(out: &mut String, name: &str, help: &str, value: impl std::fmt::Display) { out.push_str("# HELP "); out.push_str(name); @@ -413,17 +423,8 @@ impl HealthMachine { fn apply_boot(&mut self, event: HealthEvent, now: Instant) { match event { - HealthEvent::ProcessStarted => { - if !self.started { - self.started = true; - self.started_at = Some(now); - } - } - HealthEvent::InitializationCompleted => { - if self.fatal.is_none() && !self.draining { - self.initialized = true; - } - } + HealthEvent::ProcessStarted => self.mark_started(now), + HealthEvent::InitializationCompleted => self.mark_initialized(), HealthEvent::InitializationFailed { detail } => { self.enter_fatal(FatalCode::InitializationFailed, detail); } @@ -431,6 +432,19 @@ impl HealthMachine { } } + fn mark_started(&mut self, now: Instant) { + if !self.started { + self.started = true; + self.started_at = Some(now); + } + } + + fn mark_initialized(&mut self) { + if self.fatal.is_none() && !self.draining { + self.initialized = true; + } + } + fn apply_checkpoint(&mut self, event: HealthEvent, now: Instant) { match event { HealthEvent::CheckpointValidated { identity } => { @@ -486,24 +500,10 @@ impl HealthMachine { let now = self.clock.now(); let origin = self.started_at.unwrap_or(now); let ms = |t: Instant| duration_ms(t.saturating_duration_since(origin)); - let age_base = self.last_ingress.or(self.checkpoint_at); - let age_ms = age_base.map(|t| duration_ms(now.saturating_duration_since(t))); let stale = self.compute_stale(now); - let live = self.started; - let ready = live - && self.initialized - && self.checkpoint_ok - && !self.draining - && self.fatal.is_none(); - let ratio = if self.queue_capacity == 0 { - None - } else { - Some(self.queue_depth as f64 / self.queue_capacity as f64) - }; - HealthSnapshot { - live, - ready, + live: self.started, + ready: self.is_ready(), phase: self.phase(stale), reasons: self.reasons(stale), last_successful_tick_ms: self.last_tick.map(ms), @@ -511,28 +511,71 @@ impl HealthMachine { .last_tick .map(|t| duration_ms(now.saturating_duration_since(t))), checkpoint: self.checkpoint.clone(), - input_freshness: InputFreshness { age_ms, stale }, - queue_pressure: QueuePressure { - depth: self.queue_depth, - capacity: self.queue_capacity, - ratio, - overloaded: self.overloaded, + input_freshness: InputFreshness { + age_ms: self.input_age_ms(now), + stale, }, + queue_pressure: self.queue_pressure(), fatal: self.fatal.clone(), observed_at_ms: ms(now), } } + fn is_ready(&self) -> bool { + self.is_initialized() && self.is_serving() + } + + fn is_initialized(&self) -> bool { + self.started && self.initialized && self.checkpoint_ok + } + + fn is_serving(&self) -> bool { + !self.draining && self.fatal.is_none() + } + + fn input_age_ms(&self, now: Instant) -> Option { + self.last_ingress + .or(self.checkpoint_at) + .map(|t| duration_ms(now.saturating_duration_since(t))) + } + + fn queue_pressure(&self) -> QueuePressure { + QueuePressure { + depth: self.queue_depth, + capacity: self.queue_capacity, + ratio: self.queue_ratio(), + overloaded: self.overloaded, + } + } + + fn queue_ratio(&self) -> Option { + if self.queue_capacity == 0 { + None + } else { + Some(self.queue_depth as f64 / self.queue_capacity as f64) + } + } + fn reasons(&self, stale: bool) -> Vec { if self.fatal.is_some() { return vec![ReasonCode::Fatal]; } - let mut reasons = Vec::default(); + let mut reasons = self.boot_reasons(); + self.push_runtime_reasons(&mut reasons, stale); + reasons + } + + fn boot_reasons(&self) -> Vec { if !self.initialized { - reasons.push(ReasonCode::Starting); + vec![ReasonCode::Starting] } else if !self.checkpoint_ok { - reasons.push(ReasonCode::CheckpointPending); + vec![ReasonCode::CheckpointPending] + } else { + Vec::default() } + } + + fn push_runtime_reasons(&self, reasons: &mut Vec, stale: bool) { if self.draining { reasons.push(ReasonCode::Draining); } @@ -542,17 +585,33 @@ impl HealthMachine { if self.overloaded { reasons.push(ReasonCode::Overload); } - reasons } fn phase(&self, stale: bool) -> HealthPhase { + self.terminal_phase() + .unwrap_or_else(|| self.operational_phase(stale)) + } + + fn terminal_phase(&self) -> Option { if self.fatal.is_some() { - HealthPhase::Fatal + Some(HealthPhase::Fatal) } else if self.draining { - HealthPhase::Draining - } else if !self.initialized { + Some(HealthPhase::Draining) + } else { + None + } + } + + fn operational_phase(&self, stale: bool) -> HealthPhase { + if !self.initialized { HealthPhase::Starting - } else if !self.checkpoint_ok { + } else { + self.checkpoint_phase(stale) + } + } + + fn checkpoint_phase(&self, stale: bool) -> HealthPhase { + if !self.checkpoint_ok { HealthPhase::LoadingCheckpoint } else if stale || self.overloaded { HealthPhase::Degraded From d23c38f23211a5d6ecb45214d2d96692cd1f6ff0 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:16:05 +0000 Subject: [PATCH 07/18] fix: split health module so Codacy file-length stays under 500 Codacy's remaining medium issue was file NLOC on src/health/mod.rs (596). Move clock, snapshot/metrics, state machine, and handle into sibling modules and keep the public health API via re-exports. Co-authored-by: Raul Cardenas Montoya --- CHANGELOG.md | 2 +- docs/health.md | 4 +- src/health/clock.rs | 54 ++++ src/health/handle.rs | 52 +++ src/health/machine.rs | 366 +++++++++++++++++++++ src/health/mod.rs | 700 +---------------------------------------- src/health/snapshot.rs | 240 ++++++++++++++ src/health/tests.rs | 2 +- 8 files changed, 728 insertions(+), 692 deletions(-) create mode 100644 src/health/clock.rs create mode 100644 src/health/handle.rs create mode 100644 src/health/machine.rs create mode 100644 src/health/snapshot.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index c5d6add..7fbf69c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,7 +10,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added - Distinct liveness, readiness, recoverable degradation, and sticky fatal health - (`src/health.rs`) with a fake-clock state-machine test for every transition and + (`src/health/`) with a fake-clock state-machine test for every transition and recovery path. Optional `control_bind` listener serves `/livez`, `/readyz`, `/health`, and `/metrics` (the repository's first control surface; none existed before). Contract, transition table, and example snapshots: [`docs/health.md`](docs/health.md). diff --git a/docs/health.md b/docs/health.md index 1179b5b..bc01a55 100644 --- a/docs/health.md +++ b/docs/health.md @@ -11,13 +11,13 @@ This repository had no HTTP/metrics server before this surface. When |---|---| | `GET /livez` | `200` if live, `503` otherwise | | `GET /readyz` | `200` if ready, `503` otherwise | -| `GET /health` | `200` JSON [`HealthSnapshot`](../src/health.rs) (always; inspect `phase`) | +| `GET /health` | `200` JSON [`HealthSnapshot`](../src/health/snapshot.rs) (always; inspect `phase`) | | `GET /metrics` | Prometheus text; labels are phase/reason codes only | Leave `control_bind` unset to preserve the historical no-extra-socket default. Do not add a second control server beside this one. -Library embedders can also clone [`HealthHandle`](../src/health.rs) from +Library embedders can also clone [`HealthHandle`](../src/health/handle.rs) from `BrainstemDaemon::health()` and call `snapshot()` / `try_snapshot()` without waiting on the tick loop's backend or `SpikingNetwork::step`. diff --git a/src/health/clock.rs b/src/health/clock.rs new file mode 100644 index 0000000..03bf2a0 --- /dev/null +++ b/src/health/clock.rs @@ -0,0 +1,54 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::{Duration, Instant}; + +/// Monotonic clock used by the health state machine. +pub trait Clock: Send + Sync { + fn now(&self) -> Instant; +} + +/// Wall-clock monotonic clock. +#[derive(Debug, Clone, Copy, Default)] +pub struct SystemClock; + +impl Clock for SystemClock { + fn now(&self) -> Instant { + Instant::now() + } +} + +/// Test clock. Clone and share the same offset with a [`super::HealthMachine`]. +#[derive(Debug, Clone)] +pub struct FakeClock { + origin: Instant, + offset_nanos: Arc, +} + +impl FakeClock { + pub fn new() -> Self { + Self::default() + } + + pub fn advance(&self, duration: Duration) { + let add = u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX); + self.offset_nanos.fetch_add(add, Ordering::SeqCst); + } +} + +impl Default for FakeClock { + fn default() -> Self { + Self { + origin: Instant::now(), + offset_nanos: Arc::new(AtomicU64::new(0)), + } + } +} + +impl Clock for FakeClock { + fn now(&self) -> Instant { + self.origin + Duration::from_nanos(self.offset_nanos.load(Ordering::SeqCst)) + } +} diff --git a/src/health/handle.rs b/src/health/handle.rs new file mode 100644 index 0000000..12024c3 --- /dev/null +++ b/src/health/handle.rs @@ -0,0 +1,52 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +use std::sync::{Arc, RwLock}; + +use super::clock::SystemClock; +use super::machine::{HealthEvent, HealthLimits, HealthMachine}; +use super::snapshot::HealthSnapshot; + +/// Cloneable, non-tick-blocking handle for supervisors and the control surface. +#[derive(Clone)] +pub struct HealthHandle { + inner: Arc>, +} + +impl HealthHandle { + /// Live process, not yet ready. Used when the daemon is constructed. + pub fn started(limits: HealthLimits) -> Self { + let mut machine = HealthMachine::new(SystemClock, limits); + machine.apply(HealthEvent::ProcessStarted); + Self { + inner: Arc::new(RwLock::new(machine)), + } + } + + pub fn from_machine(machine: HealthMachine) -> Self { + Self { + inner: Arc::new(RwLock::new(machine)), + } + } + + pub fn apply(&self, event: HealthEvent) { + let mut guard = self.inner.write().unwrap_or_else(|e| e.into_inner()); + guard.apply(event); + } + + /// Clone the current snapshot. May wait only for an in-flight `apply` (no I/O). + pub fn snapshot(&self) -> HealthSnapshot { + let guard = self.inner.read().unwrap_or_else(|e| e.into_inner()); + guard.snapshot() + } + + /// Never waits. Returns `None` if a writer currently holds the lock. + pub fn try_snapshot(&self) -> Option { + self.inner.try_read().ok().map(|guard| guard.snapshot()) + } + + #[cfg(test)] + pub(crate) fn lock_write_for_test(&self) -> std::sync::RwLockWriteGuard<'_, HealthMachine> { + self.inner.write().unwrap_or_else(|e| e.into_inner()) + } +} diff --git a/src/health/machine.rs b/src/health/machine.rs new file mode 100644 index 0000000..8a08c90 --- /dev/null +++ b/src/health/machine.rs @@ -0,0 +1,366 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use super::clock::Clock; +use super::snapshot::{ + CheckpointIdentity, FatalCode, FatalState, HealthPhase, HealthSnapshot, InputFreshness, + QueuePressure, ReasonCode, +}; + +/// Thresholds for recoverable degradation. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct HealthLimits { + /// Ingress older than this marks `stale_input`. + pub stale_after: Duration, + /// Queue fill ratio that *enters* overload (`depth / capacity`). + pub overload_high: f64, + /// Queue fill ratio that *exits* overload (hysteresis; must be `<= overload_high`). + pub overload_low: f64, +} + +impl Default for HealthLimits { + fn default() -> Self { + Self { + stale_after: Duration::from_millis(500), + overload_high: 0.90, + overload_low: 0.70, + } + } +} + +impl HealthLimits { + /// Replace non-finite or inverted watermarks with the built-in defaults. + pub fn sanitized(self) -> Self { + let high = finite_or(self.overload_high, 0.90); + let low = finite_or(self.overload_low, 0.70); + if high >= low { + Self { + stale_after: self.stale_after, + overload_high: high, + overload_low: low, + } + } else { + Self::default() + } + } +} + +fn finite_or(value: f64, fallback: f64) -> f64 { + if value.is_finite() { value } else { fallback } +} + +/// State-machine events. Tick-loop I/O never runs while these are applied. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum HealthEvent { + ProcessStarted, + InitializationCompleted, + InitializationFailed { detail: String }, + CheckpointValidated { identity: CheckpointIdentity }, + CheckpointRejected { detail: String }, + TickSucceeded, + IngressObserved, + QueuePressure { depth: u64, capacity: u64 }, + BeginDrain, + Fatal { code: FatalCode, detail: String }, +} + +/// Pure health state machine. Drive it with [`super::FakeClock`] in tests. +pub struct HealthMachine { + clock: Arc, + limits: HealthLimits, + started: bool, + initialized: bool, + checkpoint_ok: bool, + draining: bool, + started_at: Option, + checkpoint_at: Option, + last_tick: Option, + last_ingress: Option, + checkpoint: Option, + queue_depth: u64, + queue_capacity: u64, + overloaded: bool, + fatal: Option, +} + +impl HealthMachine { + pub fn new(clock: impl Clock + 'static, limits: HealthLimits) -> Self { + Self { + clock: Arc::new(clock), + limits: limits.sanitized(), + started: false, + initialized: false, + checkpoint_ok: false, + draining: false, + started_at: None, + checkpoint_at: None, + last_tick: None, + last_ingress: None, + checkpoint: None, + queue_depth: 0, + queue_capacity: 0, + overloaded: false, + fatal: None, + } + } + + pub fn apply(&mut self, event: HealthEvent) { + let now = self.clock.now(); + match event { + HealthEvent::TickSucceeded + | HealthEvent::IngressObserved + | HealthEvent::QueuePressure { .. } => self.apply_runtime(event, now), + other => self.apply_lifecycle(other, now), + } + } + + fn apply_lifecycle(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::ProcessStarted + | HealthEvent::InitializationCompleted + | HealthEvent::InitializationFailed { .. } => self.apply_boot(event, now), + HealthEvent::CheckpointValidated { .. } | HealthEvent::CheckpointRejected { .. } => { + self.apply_checkpoint(event, now) + } + HealthEvent::BeginDrain | HealthEvent::Fatal { .. } => self.apply_terminal(event), + _ => {} + } + } + + fn apply_boot(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::ProcessStarted => self.mark_started(now), + HealthEvent::InitializationCompleted => self.mark_initialized(), + HealthEvent::InitializationFailed { detail } => { + self.enter_fatal(FatalCode::InitializationFailed, detail); + } + _ => {} + } + } + + fn mark_started(&mut self, now: Instant) { + if !self.started { + self.started = true; + self.started_at = Some(now); + } + } + + fn mark_initialized(&mut self) { + if self.fatal.is_none() && !self.draining { + self.initialized = true; + } + } + + fn apply_checkpoint(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::CheckpointValidated { identity } => { + if self.can_accept_checkpoint() { + self.checkpoint = Some(identity); + self.checkpoint_ok = true; + self.checkpoint_at = Some(now); + } + } + HealthEvent::CheckpointRejected { detail } => { + self.enter_fatal(FatalCode::CheckpointInvalid, detail); + } + _ => {} + } + } + + fn can_accept_checkpoint(&self) -> bool { + self.fatal.is_none() && !self.draining && self.initialized + } + + fn apply_terminal(&mut self, event: HealthEvent) { + match event { + HealthEvent::BeginDrain => { + if self.fatal.is_none() { + self.draining = true; + } + } + HealthEvent::Fatal { code, detail } => self.enter_fatal(code, detail), + _ => {} + } + } + + fn apply_runtime(&mut self, event: HealthEvent, now: Instant) { + match event { + HealthEvent::TickSucceeded => self.last_tick = Some(now), + HealthEvent::IngressObserved => self.last_ingress = Some(now), + HealthEvent::QueuePressure { depth, capacity } => { + self.queue_depth = depth; + self.queue_capacity = capacity; + self.overloaded = next_overload( + self.overloaded, + depth, + capacity, + self.limits.overload_high, + self.limits.overload_low, + ); + } + _ => {} + } + } + + pub fn snapshot(&self) -> HealthSnapshot { + let now = self.clock.now(); + let origin = self.started_at.unwrap_or(now); + let ms = |t: Instant| duration_ms(t.saturating_duration_since(origin)); + let stale = self.compute_stale(now); + HealthSnapshot { + live: self.started, + ready: self.is_ready(), + phase: self.phase(stale), + reasons: self.reasons(stale), + last_successful_tick_ms: self.last_tick.map(ms), + tick_age_ms: self + .last_tick + .map(|t| duration_ms(now.saturating_duration_since(t))), + checkpoint: self.checkpoint.clone(), + input_freshness: InputFreshness { + age_ms: self.input_age_ms(now), + stale, + }, + queue_pressure: self.queue_pressure(), + fatal: self.fatal.clone(), + observed_at_ms: ms(now), + } + } + + fn is_ready(&self) -> bool { + self.is_initialized() && self.is_serving() + } + + fn is_initialized(&self) -> bool { + self.started && self.initialized && self.checkpoint_ok + } + + fn is_serving(&self) -> bool { + !self.draining && self.fatal.is_none() + } + + fn input_age_ms(&self, now: Instant) -> Option { + self.last_ingress + .or(self.checkpoint_at) + .map(|t| duration_ms(now.saturating_duration_since(t))) + } + + fn queue_pressure(&self) -> QueuePressure { + QueuePressure { + depth: self.queue_depth, + capacity: self.queue_capacity, + ratio: self.queue_ratio(), + overloaded: self.overloaded, + } + } + + fn queue_ratio(&self) -> Option { + if self.queue_capacity == 0 { + None + } else { + Some(self.queue_depth as f64 / self.queue_capacity as f64) + } + } + + fn reasons(&self, stale: bool) -> Vec { + if self.fatal.is_some() { + return vec![ReasonCode::Fatal]; + } + let mut reasons = self.boot_reasons(); + self.push_runtime_reasons(&mut reasons, stale); + reasons + } + + fn boot_reasons(&self) -> Vec { + if !self.initialized { + vec![ReasonCode::Starting] + } else if !self.checkpoint_ok { + vec![ReasonCode::CheckpointPending] + } else { + Vec::default() + } + } + + fn push_runtime_reasons(&self, reasons: &mut Vec, stale: bool) { + if self.draining { + reasons.push(ReasonCode::Draining); + } + if stale { + reasons.push(ReasonCode::StaleInput); + } + if self.overloaded { + reasons.push(ReasonCode::Overload); + } + } + + fn phase(&self, stale: bool) -> HealthPhase { + self.terminal_phase() + .unwrap_or_else(|| self.operational_phase(stale)) + } + + fn terminal_phase(&self) -> Option { + if self.fatal.is_some() { + Some(HealthPhase::Fatal) + } else if self.draining { + Some(HealthPhase::Draining) + } else { + None + } + } + + fn operational_phase(&self, stale: bool) -> HealthPhase { + if !self.initialized { + HealthPhase::Starting + } else { + self.checkpoint_phase(stale) + } + } + + fn checkpoint_phase(&self, stale: bool) -> HealthPhase { + if !self.checkpoint_ok { + HealthPhase::LoadingCheckpoint + } else if stale || self.overloaded { + HealthPhase::Degraded + } else { + HealthPhase::Running + } + } + + fn enter_fatal(&mut self, code: FatalCode, detail: String) { + if self.fatal.is_some() { + return; + } + self.fatal = Some(FatalState { code, detail }); + self.checkpoint_ok = false; + } + + fn compute_stale(&self, now: Instant) -> bool { + if !self.initialized || !self.checkpoint_ok { + return false; + } + let baseline = match self.last_ingress.or(self.checkpoint_at) { + Some(t) => t, + None => return false, + }; + now.saturating_duration_since(baseline) >= self.limits.stale_after + } +} + +fn next_overload(currently: bool, depth: u64, capacity: u64, high: f64, low: f64) -> bool { + if capacity == 0 { + return false; + } + let ratio = depth as f64 / capacity as f64; + if currently { + ratio > low + } else { + ratio >= high + } +} + +fn duration_ms(d: Duration) -> u64 { + u64::try_from(d.as_millis()).unwrap_or(u64::MAX) +} diff --git a/src/health/mod.rs b/src/health/mod.rs index ce8161a..fcbab14 100644 --- a/src/health/mod.rs +++ b/src/health/mod.rs @@ -11,694 +11,18 @@ //! Snapshot reads take a separate lock from the tick loop's backend and network, so they //! do not wait on `StimulusSource` or `SpikingNetwork::step`. -use std::sync::atomic::{AtomicU64, Ordering}; -use std::sync::{Arc, RwLock}; -use std::time::{Duration, Instant}; - -use serde::{Deserialize, Serialize}; - -/// Monotonic clock used by the health state machine. -pub trait Clock: Send + Sync { - fn now(&self) -> Instant; -} - -/// Wall-clock monotonic clock. -#[derive(Debug, Clone, Copy, Default)] -pub struct SystemClock; - -impl Clock for SystemClock { - fn now(&self) -> Instant { - Instant::now() - } -} - -/// Test clock. Clone and share the same offset with a [`HealthMachine`]. -#[derive(Debug, Clone)] -pub struct FakeClock { - origin: Instant, - offset_nanos: Arc, -} - -impl FakeClock { - pub fn new() -> Self { - Self::default() - } - - pub fn advance(&self, duration: Duration) { - let add = u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX); - self.offset_nanos.fetch_add(add, Ordering::SeqCst); - } -} - -impl Default for FakeClock { - fn default() -> Self { - Self { - origin: Instant::now(), - offset_nanos: Arc::new(AtomicU64::new(0)), - } - } -} - -impl Clock for FakeClock { - fn now(&self) -> Instant { - self.origin + Duration::from_nanos(self.offset_nanos.load(Ordering::SeqCst)) - } -} - -/// Thresholds for recoverable degradation. -#[derive(Debug, Clone, Copy, PartialEq)] -pub struct HealthLimits { - /// Ingress older than this marks `stale_input`. - pub stale_after: Duration, - /// Queue fill ratio that *enters* overload (`depth / capacity`). - pub overload_high: f64, - /// Queue fill ratio that *exits* overload (hysteresis; must be `<= overload_high`). - pub overload_low: f64, -} - -impl Default for HealthLimits { - fn default() -> Self { - Self { - stale_after: Duration::from_millis(500), - overload_high: 0.90, - overload_low: 0.70, - } - } -} - -impl HealthLimits { - /// Replace non-finite or inverted watermarks with the built-in defaults. - pub fn sanitized(self) -> Self { - let high = finite_or(self.overload_high, 0.90); - let low = finite_or(self.overload_low, 0.70); - if high >= low { - Self { - stale_after: self.stale_after, - overload_high: high, - overload_low: low, - } - } else { - Self::default() - } - } -} - -fn finite_or(value: f64, fallback: f64) -> f64 { - if value.is_finite() { value } else { fallback } -} - -/// Coarse phase derived from the snapshot. Stable, low-cardinality. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum HealthPhase { - Starting, - LoadingCheckpoint, - Running, - Degraded, - Draining, - Fatal, -} - -impl HealthPhase { - pub const ALL: [HealthPhase; 6] = [ - HealthPhase::Starting, - HealthPhase::LoadingCheckpoint, - HealthPhase::Running, - HealthPhase::Degraded, - HealthPhase::Draining, - HealthPhase::Fatal, - ]; - - pub fn as_str(self) -> &'static str { - match self { - HealthPhase::Starting => "starting", - HealthPhase::LoadingCheckpoint => "loading_checkpoint", - HealthPhase::Running => "running", - HealthPhase::Degraded => "degraded", - HealthPhase::Draining => "draining", - HealthPhase::Fatal => "fatal", - } - } -} - -/// Stable reason codes. Never put detailed error text in metric labels — use these codes. -#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum ReasonCode { - Starting, - CheckpointPending, - StaleInput, - Overload, - Draining, - Fatal, -} - -impl ReasonCode { - pub const RECOVERABLE: [ReasonCode; 2] = [ReasonCode::StaleInput, ReasonCode::Overload]; - - pub fn as_str(self) -> &'static str { - match self { - ReasonCode::Starting => "starting", - ReasonCode::CheckpointPending => "checkpoint_pending", - ReasonCode::StaleInput => "stale_input", - ReasonCode::Overload => "overload", - ReasonCode::Draining => "draining", - ReasonCode::Fatal => "fatal", - } - } -} - -/// Sticky fatal class. Low-cardinality; details live on [`FatalState::detail`]. -#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] -#[serde(rename_all = "snake_case")] -pub enum FatalCode { - InitializationFailed, - CheckpointInvalid, - Unspecified, -} - -impl FatalCode { - pub fn as_str(self) -> &'static str { - match self { - FatalCode::InitializationFailed => "initialization_failed", - FatalCode::CheckpointInvalid => "checkpoint_invalid", - FatalCode::Unspecified => "unspecified", - } - } -} - -/// Identity of the loaded checkpoint. Digest may be absent until real validation lands. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct CheckpointIdentity { - pub id: String, - pub digest: Option, -} - -/// Fatal snapshot payload. `detail` is for logs/JSON, never a Prometheus label. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct FatalState { - pub code: FatalCode, - pub detail: String, -} - -/// Ingress freshness relative to the health clock. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -pub struct InputFreshness { - pub age_ms: Option, - pub stale: bool, -} - -/// Bounded-queue pressure. `capacity == 0` means "no queue instrumented". -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct QueuePressure { - pub depth: u64, - pub capacity: u64, - pub ratio: Option, - pub overloaded: bool, -} - -/// Machine-readable health view for supervisors and the control surface. -#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct HealthSnapshot { - pub live: bool, - pub ready: bool, - pub phase: HealthPhase, - pub reasons: Vec, - pub last_successful_tick_ms: Option, - /// Milliseconds since the last successful tick (`None` if none yet). - pub tick_age_ms: Option, - pub checkpoint: Option, - pub input_freshness: InputFreshness, - pub queue_pressure: QueuePressure, - pub fatal: Option, - pub observed_at_ms: u64, -} - -impl HealthSnapshot { - /// Prometheus text exposition. Labels are stable reason/phase codes only. - pub fn prometheus_text(&self) -> String { - let mut out = String::default(); - write_probe_gauges(&mut out, self); - write_labeled_gauges(&mut out, self); - write_runtime_gauges(&mut out, self); - out - } -} - -fn write_probe_gauges(out: &mut String, snap: &HealthSnapshot) { - push_gauge( - out, - "brainstem_live", - "1 if the process health reporter is running.", - u8::from(snap.live), - ); - push_gauge( - out, - "brainstem_ready", - "1 if initialized, checkpoint-valid, not draining, not fatal.", - u8::from(snap.ready), - ); -} - -fn write_labeled_gauges(out: &mut String, snap: &HealthSnapshot) { - out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); - out.push_str("# TYPE brainstem_phase gauge\n"); - write_phase_gauges(out, snap.phase); - out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); - out.push_str("# TYPE brainstem_degraded gauge\n"); - write_degraded_gauges(out, &snap.reasons); -} - -fn write_runtime_gauges(out: &mut String, snap: &HealthSnapshot) { - push_gauge( - out, - "brainstem_fatal", - "1 if this process has entered a sticky fatal state.", - u8::from(snap.fatal.is_some()), - ); - push_gauge( - out, - "brainstem_last_successful_tick_ms", - "Milliseconds from process start until the last successful tick (not tick age).", - snap.last_successful_tick_ms.unwrap_or(0), - ); - push_gauge( - out, - "brainstem_tick_age_ms", - "Milliseconds since the last successful tick; 0 if none yet.", - snap.tick_age_ms.unwrap_or(0), - ); - push_gauge( - out, - "brainstem_input_age_ms", - "Age of last ingress (or checkpoint, if none) in milliseconds.", - snap.input_freshness.age_ms.unwrap_or(0), - ); - push_gauge( - out, - "brainstem_queue_depth", - "Ingress queue depth.", - snap.queue_pressure.depth, - ); - push_gauge( - out, - "brainstem_queue_capacity", - "Ingress queue capacity.", - snap.queue_pressure.capacity, - ); -} - -fn push_gauge(out: &mut String, name: &str, help: &str, value: impl std::fmt::Display) { - out.push_str("# HELP "); - out.push_str(name); - out.push(' '); - out.push_str(help); - out.push('\n'); - out.push_str("# TYPE "); - out.push_str(name); - out.push_str(" gauge\n"); - out.push_str(name); - out.push(' '); - out.push_str(&value.to_string()); - out.push('\n'); -} - -fn write_phase_gauges(out: &mut String, current: HealthPhase) { - for phase in HealthPhase::ALL { - out.push_str(&format!( - "brainstem_phase{{phase=\"{}\"}} {}\n", - phase.as_str(), - u8::from(current == phase) - )); - } -} - -fn write_degraded_gauges(out: &mut String, reasons: &[ReasonCode]) { - for reason in ReasonCode::RECOVERABLE { - out.push_str(&format!( - "brainstem_degraded{{reason=\"{}\"}} {}\n", - reason.as_str(), - u8::from(reasons.contains(&reason)) - )); - } -} - -/// State-machine events. Tick-loop I/O never runs while these are applied. -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum HealthEvent { - ProcessStarted, - InitializationCompleted, - InitializationFailed { detail: String }, - CheckpointValidated { identity: CheckpointIdentity }, - CheckpointRejected { detail: String }, - TickSucceeded, - IngressObserved, - QueuePressure { depth: u64, capacity: u64 }, - BeginDrain, - Fatal { code: FatalCode, detail: String }, -} - -/// Pure health state machine. Drive it with [`FakeClock`] in tests. -pub struct HealthMachine { - clock: Arc, - limits: HealthLimits, - started: bool, - initialized: bool, - checkpoint_ok: bool, - draining: bool, - started_at: Option, - checkpoint_at: Option, - last_tick: Option, - last_ingress: Option, - checkpoint: Option, - queue_depth: u64, - queue_capacity: u64, - overloaded: bool, - fatal: Option, -} - -impl HealthMachine { - pub fn new(clock: impl Clock + 'static, limits: HealthLimits) -> Self { - Self { - clock: Arc::new(clock), - limits: limits.sanitized(), - started: false, - initialized: false, - checkpoint_ok: false, - draining: false, - started_at: None, - checkpoint_at: None, - last_tick: None, - last_ingress: None, - checkpoint: None, - queue_depth: 0, - queue_capacity: 0, - overloaded: false, - fatal: None, - } - } - - pub fn apply(&mut self, event: HealthEvent) { - let now = self.clock.now(); - match event { - HealthEvent::TickSucceeded - | HealthEvent::IngressObserved - | HealthEvent::QueuePressure { .. } => self.apply_runtime(event, now), - other => self.apply_lifecycle(other, now), - } - } - - fn apply_lifecycle(&mut self, event: HealthEvent, now: Instant) { - match event { - HealthEvent::ProcessStarted - | HealthEvent::InitializationCompleted - | HealthEvent::InitializationFailed { .. } => self.apply_boot(event, now), - HealthEvent::CheckpointValidated { .. } | HealthEvent::CheckpointRejected { .. } => { - self.apply_checkpoint(event, now) - } - HealthEvent::BeginDrain | HealthEvent::Fatal { .. } => self.apply_terminal(event), - _ => {} - } - } - - fn apply_boot(&mut self, event: HealthEvent, now: Instant) { - match event { - HealthEvent::ProcessStarted => self.mark_started(now), - HealthEvent::InitializationCompleted => self.mark_initialized(), - HealthEvent::InitializationFailed { detail } => { - self.enter_fatal(FatalCode::InitializationFailed, detail); - } - _ => {} - } - } - - fn mark_started(&mut self, now: Instant) { - if !self.started { - self.started = true; - self.started_at = Some(now); - } - } - - fn mark_initialized(&mut self) { - if self.fatal.is_none() && !self.draining { - self.initialized = true; - } - } - - fn apply_checkpoint(&mut self, event: HealthEvent, now: Instant) { - match event { - HealthEvent::CheckpointValidated { identity } => { - if self.can_accept_checkpoint() { - self.checkpoint = Some(identity); - self.checkpoint_ok = true; - self.checkpoint_at = Some(now); - } - } - HealthEvent::CheckpointRejected { detail } => { - self.enter_fatal(FatalCode::CheckpointInvalid, detail); - } - _ => {} - } - } - - fn can_accept_checkpoint(&self) -> bool { - self.fatal.is_none() && !self.draining && self.initialized - } - - fn apply_terminal(&mut self, event: HealthEvent) { - match event { - HealthEvent::BeginDrain => { - if self.fatal.is_none() { - self.draining = true; - } - } - HealthEvent::Fatal { code, detail } => self.enter_fatal(code, detail), - _ => {} - } - } - - fn apply_runtime(&mut self, event: HealthEvent, now: Instant) { - match event { - HealthEvent::TickSucceeded => self.last_tick = Some(now), - HealthEvent::IngressObserved => self.last_ingress = Some(now), - HealthEvent::QueuePressure { depth, capacity } => { - self.queue_depth = depth; - self.queue_capacity = capacity; - self.overloaded = next_overload( - self.overloaded, - depth, - capacity, - self.limits.overload_high, - self.limits.overload_low, - ); - } - _ => {} - } - } - - pub fn snapshot(&self) -> HealthSnapshot { - let now = self.clock.now(); - let origin = self.started_at.unwrap_or(now); - let ms = |t: Instant| duration_ms(t.saturating_duration_since(origin)); - let stale = self.compute_stale(now); - HealthSnapshot { - live: self.started, - ready: self.is_ready(), - phase: self.phase(stale), - reasons: self.reasons(stale), - last_successful_tick_ms: self.last_tick.map(ms), - tick_age_ms: self - .last_tick - .map(|t| duration_ms(now.saturating_duration_since(t))), - checkpoint: self.checkpoint.clone(), - input_freshness: InputFreshness { - age_ms: self.input_age_ms(now), - stale, - }, - queue_pressure: self.queue_pressure(), - fatal: self.fatal.clone(), - observed_at_ms: ms(now), - } - } - - fn is_ready(&self) -> bool { - self.is_initialized() && self.is_serving() - } - - fn is_initialized(&self) -> bool { - self.started && self.initialized && self.checkpoint_ok - } - - fn is_serving(&self) -> bool { - !self.draining && self.fatal.is_none() - } - - fn input_age_ms(&self, now: Instant) -> Option { - self.last_ingress - .or(self.checkpoint_at) - .map(|t| duration_ms(now.saturating_duration_since(t))) - } - - fn queue_pressure(&self) -> QueuePressure { - QueuePressure { - depth: self.queue_depth, - capacity: self.queue_capacity, - ratio: self.queue_ratio(), - overloaded: self.overloaded, - } - } - - fn queue_ratio(&self) -> Option { - if self.queue_capacity == 0 { - None - } else { - Some(self.queue_depth as f64 / self.queue_capacity as f64) - } - } - - fn reasons(&self, stale: bool) -> Vec { - if self.fatal.is_some() { - return vec![ReasonCode::Fatal]; - } - let mut reasons = self.boot_reasons(); - self.push_runtime_reasons(&mut reasons, stale); - reasons - } - - fn boot_reasons(&self) -> Vec { - if !self.initialized { - vec![ReasonCode::Starting] - } else if !self.checkpoint_ok { - vec![ReasonCode::CheckpointPending] - } else { - Vec::default() - } - } - - fn push_runtime_reasons(&self, reasons: &mut Vec, stale: bool) { - if self.draining { - reasons.push(ReasonCode::Draining); - } - if stale { - reasons.push(ReasonCode::StaleInput); - } - if self.overloaded { - reasons.push(ReasonCode::Overload); - } - } - - fn phase(&self, stale: bool) -> HealthPhase { - self.terminal_phase() - .unwrap_or_else(|| self.operational_phase(stale)) - } - - fn terminal_phase(&self) -> Option { - if self.fatal.is_some() { - Some(HealthPhase::Fatal) - } else if self.draining { - Some(HealthPhase::Draining) - } else { - None - } - } - - fn operational_phase(&self, stale: bool) -> HealthPhase { - if !self.initialized { - HealthPhase::Starting - } else { - self.checkpoint_phase(stale) - } - } - - fn checkpoint_phase(&self, stale: bool) -> HealthPhase { - if !self.checkpoint_ok { - HealthPhase::LoadingCheckpoint - } else if stale || self.overloaded { - HealthPhase::Degraded - } else { - HealthPhase::Running - } - } - - fn enter_fatal(&mut self, code: FatalCode, detail: String) { - if self.fatal.is_some() { - return; - } - self.fatal = Some(FatalState { code, detail }); - self.checkpoint_ok = false; - } - - fn compute_stale(&self, now: Instant) -> bool { - if !self.initialized || !self.checkpoint_ok { - return false; - } - let baseline = match self.last_ingress.or(self.checkpoint_at) { - Some(t) => t, - None => return false, - }; - now.saturating_duration_since(baseline) >= self.limits.stale_after - } -} - -fn next_overload(currently: bool, depth: u64, capacity: u64, high: f64, low: f64) -> bool { - if capacity == 0 { - return false; - } - let ratio = depth as f64 / capacity as f64; - if currently { - ratio > low - } else { - ratio >= high - } -} - -fn duration_ms(d: Duration) -> u64 { - u64::try_from(d.as_millis()).unwrap_or(u64::MAX) -} - -/// Cloneable, non-tick-blocking handle for supervisors and the control surface. -#[derive(Clone)] -pub struct HealthHandle { - inner: Arc>, -} - -impl HealthHandle { - /// Live process, not yet ready. Used when the daemon is constructed. - pub fn started(limits: HealthLimits) -> Self { - let mut machine = HealthMachine::new(SystemClock, limits); - machine.apply(HealthEvent::ProcessStarted); - Self { - inner: Arc::new(RwLock::new(machine)), - } - } - - pub fn from_machine(machine: HealthMachine) -> Self { - Self { - inner: Arc::new(RwLock::new(machine)), - } - } - - pub fn apply(&self, event: HealthEvent) { - let mut guard = self.inner.write().unwrap_or_else(|e| e.into_inner()); - guard.apply(event); - } - - /// Clone the current snapshot. May wait only for an in-flight `apply` (no I/O). - pub fn snapshot(&self) -> HealthSnapshot { - let guard = self.inner.read().unwrap_or_else(|e| e.into_inner()); - guard.snapshot() - } - - /// Never waits. Returns `None` if a writer currently holds the lock. - pub fn try_snapshot(&self) -> Option { - self.inner.try_read().ok().map(|guard| guard.snapshot()) - } - - #[cfg(test)] - pub(crate) fn lock_write_for_test(&self) -> std::sync::RwLockWriteGuard<'_, HealthMachine> { - self.inner.write().unwrap_or_else(|e| e.into_inner()) - } -} +mod clock; +mod handle; +mod machine; +mod snapshot; + +pub use clock::{Clock, FakeClock, SystemClock}; +pub use handle::HealthHandle; +pub use machine::{HealthEvent, HealthLimits, HealthMachine}; +pub use snapshot::{ + CheckpointIdentity, FatalCode, FatalState, HealthPhase, HealthSnapshot, InputFreshness, + QueuePressure, ReasonCode, +}; #[cfg(test)] mod tests; diff --git a/src/health/snapshot.rs b/src/health/snapshot.rs new file mode 100644 index 0000000..1b6a415 --- /dev/null +++ b/src/health/snapshot.rs @@ -0,0 +1,240 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 +// Copyright 2026 Raul Montoya Cardenas + +use serde::{Deserialize, Serialize}; + +/// Coarse phase derived from the snapshot. Stable, low-cardinality. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum HealthPhase { + Starting, + LoadingCheckpoint, + Running, + Degraded, + Draining, + Fatal, +} + +impl HealthPhase { + pub const ALL: [HealthPhase; 6] = [ + HealthPhase::Starting, + HealthPhase::LoadingCheckpoint, + HealthPhase::Running, + HealthPhase::Degraded, + HealthPhase::Draining, + HealthPhase::Fatal, + ]; + + pub fn as_str(self) -> &'static str { + match self { + HealthPhase::Starting => "starting", + HealthPhase::LoadingCheckpoint => "loading_checkpoint", + HealthPhase::Running => "running", + HealthPhase::Degraded => "degraded", + HealthPhase::Draining => "draining", + HealthPhase::Fatal => "fatal", + } + } +} + +/// Stable reason codes. Never put detailed error text in metric labels — use these codes. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ReasonCode { + Starting, + CheckpointPending, + StaleInput, + Overload, + Draining, + Fatal, +} + +impl ReasonCode { + pub const RECOVERABLE: [ReasonCode; 2] = [ReasonCode::StaleInput, ReasonCode::Overload]; + + pub fn as_str(self) -> &'static str { + match self { + ReasonCode::Starting => "starting", + ReasonCode::CheckpointPending => "checkpoint_pending", + ReasonCode::StaleInput => "stale_input", + ReasonCode::Overload => "overload", + ReasonCode::Draining => "draining", + ReasonCode::Fatal => "fatal", + } + } +} + +/// Sticky fatal class. Low-cardinality; details live on [`FatalState::detail`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum FatalCode { + InitializationFailed, + CheckpointInvalid, + Unspecified, +} + +impl FatalCode { + pub fn as_str(self) -> &'static str { + match self { + FatalCode::InitializationFailed => "initialization_failed", + FatalCode::CheckpointInvalid => "checkpoint_invalid", + FatalCode::Unspecified => "unspecified", + } + } +} + +/// Identity of the loaded checkpoint. Digest may be absent until real validation lands. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct CheckpointIdentity { + pub id: String, + pub digest: Option, +} + +/// Fatal snapshot payload. `detail` is for logs/JSON, never a Prometheus label. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct FatalState { + pub code: FatalCode, + pub detail: String, +} + +/// Ingress freshness relative to the health clock. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct InputFreshness { + pub age_ms: Option, + pub stale: bool, +} + +/// Bounded-queue pressure. `capacity == 0` means "no queue instrumented". +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct QueuePressure { + pub depth: u64, + pub capacity: u64, + pub ratio: Option, + pub overloaded: bool, +} + +/// Machine-readable health view for supervisors and the control surface. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct HealthSnapshot { + pub live: bool, + pub ready: bool, + pub phase: HealthPhase, + pub reasons: Vec, + pub last_successful_tick_ms: Option, + /// Milliseconds since the last successful tick (`None` if none yet). + pub tick_age_ms: Option, + pub checkpoint: Option, + pub input_freshness: InputFreshness, + pub queue_pressure: QueuePressure, + pub fatal: Option, + pub observed_at_ms: u64, +} + +impl HealthSnapshot { + /// Prometheus text exposition. Labels are stable reason/phase codes only. + pub fn prometheus_text(&self) -> String { + let mut out = String::default(); + write_probe_gauges(&mut out, self); + write_labeled_gauges(&mut out, self); + write_runtime_gauges(&mut out, self); + out + } +} + +fn write_probe_gauges(out: &mut String, snap: &HealthSnapshot) { + push_gauge( + out, + "brainstem_live", + "1 if the process health reporter is running.", + u8::from(snap.live), + ); + push_gauge( + out, + "brainstem_ready", + "1 if initialized, checkpoint-valid, not draining, not fatal.", + u8::from(snap.ready), + ); +} + +fn write_labeled_gauges(out: &mut String, snap: &HealthSnapshot) { + out.push_str("# HELP brainstem_phase 1 for the current health phase.\n"); + out.push_str("# TYPE brainstem_phase gauge\n"); + write_phase_gauges(out, snap.phase); + out.push_str("# HELP brainstem_degraded Recoverable degradation by stable reason code.\n"); + out.push_str("# TYPE brainstem_degraded gauge\n"); + write_degraded_gauges(out, &snap.reasons); +} + +fn write_runtime_gauges(out: &mut String, snap: &HealthSnapshot) { + push_gauge( + out, + "brainstem_fatal", + "1 if this process has entered a sticky fatal state.", + u8::from(snap.fatal.is_some()), + ); + push_gauge( + out, + "brainstem_last_successful_tick_ms", + "Milliseconds from process start until the last successful tick (not tick age).", + snap.last_successful_tick_ms.unwrap_or(0), + ); + push_gauge( + out, + "brainstem_tick_age_ms", + "Milliseconds since the last successful tick; 0 if none yet.", + snap.tick_age_ms.unwrap_or(0), + ); + push_gauge( + out, + "brainstem_input_age_ms", + "Age of last ingress (or checkpoint, if none) in milliseconds.", + snap.input_freshness.age_ms.unwrap_or(0), + ); + push_gauge( + out, + "brainstem_queue_depth", + "Ingress queue depth.", + snap.queue_pressure.depth, + ); + push_gauge( + out, + "brainstem_queue_capacity", + "Ingress queue capacity.", + snap.queue_pressure.capacity, + ); +} + +fn push_gauge(out: &mut String, name: &str, help: &str, value: impl std::fmt::Display) { + out.push_str("# HELP "); + out.push_str(name); + out.push(' '); + out.push_str(help); + out.push('\n'); + out.push_str("# TYPE "); + out.push_str(name); + out.push_str(" gauge\n"); + out.push_str(name); + out.push(' '); + out.push_str(&value.to_string()); + out.push('\n'); +} + +fn write_phase_gauges(out: &mut String, current: HealthPhase) { + for phase in HealthPhase::ALL { + out.push_str(&format!( + "brainstem_phase{{phase=\"{}\"}} {}\n", + phase.as_str(), + u8::from(current == phase) + )); + } +} + +fn write_degraded_gauges(out: &mut String, reasons: &[ReasonCode]) { + for reason in ReasonCode::RECOVERABLE { + out.push_str(&format!( + "brainstem_degraded{{reason=\"{}\"}} {}\n", + reason.as_str(), + u8::from(reasons.contains(&reason)) + )); + } +} diff --git a/src/health/tests.rs b/src/health/tests.rs index aa158a7..5d9f361 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -1,5 +1,5 @@ use super::*; -use std::time::Duration; +use std::time::{Duration, Instant}; fn limits() -> HealthLimits { HealthLimits { From 51c90571a9753152a30c75e23e80b561695e3e17 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:23:40 +0000 Subject: [PATCH 08/18] fix: address CodeRabbit health-surface follow-ups Preserve stale_after when sanitizing inverted overload limits, recover poisoned locks in try_snapshot, back off on control accept errors, and document the initialize stand-in plus drain probe contract. Co-authored-by: Raul Cardenas Montoya --- README.md | 2 +- docs/health.md | 14 +++++++++++--- src/control.rs | 17 +++++++++++++---- src/health/handle.rs | 6 +++++- src/health/machine.rs | 6 +++++- src/health/tests.rs | 13 +++++++++++++ 6 files changed, 48 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index fb6ed08..30b0e98 100644 --- a/README.md +++ b/README.md @@ -95,7 +95,7 @@ enabled = true Default Cargo features are empty (`default = []` in `Cargo.toml`). That path uses the in-memory **stub** backend (`StubStimulusSource` + `NoopSpikeSink`) and does **not** need `libzmq`. The optional `corpus-ipc` feature (same as `--all-features` today) pulls the `corpus-ipc` git dependency and links system ZeroMQ (`libzmq3-dev` on Debian/Ubuntu). It does not vendor ZeroMQ. -`DaemonConfig` deserialization is **not** feature-gated: `spine_sub_port`, `spine_pub_port`, and `model_path` are still required in TOML even on the stub path (`services` is the only optional field, defaulting to empty). Effect at runtime depends on which backend is **wired**. +`DaemonConfig` deserialization is **not** feature-gated: `spine_sub_port`, `spine_pub_port`, and `model_path` are still required in TOML even on the stub path. Optional fields: `services` (defaults to empty) and `control_bind` (unset = no extra socket). Effect at runtime depends on which backend is **wired**. #### Feature truth table diff --git a/docs/health.md b/docs/health.md index bc01a55..04edaa3 100644 --- a/docs/health.md +++ b/docs/health.md @@ -2,7 +2,10 @@ Supervisors should treat **liveness** and **readiness** as independent. A live `brainstem-daemon` process has a health reporter; it is ready only after -stimulus-source initialization and checkpoint validation have succeeded. +`StimulusSource::initialize` succeeds. Until [`LIM-1133`](https://linear.app/rpd-34/issue/LIM-1133), +that successful `initialize` is also the checkpoint-validation stand-in: the +daemon applies `CheckpointValidated` immediately afterward. There is no +separate digest/weight check in this release. This repository had no HTTP/metrics server before this surface. When `control_bind` is set, `BrainstemDaemon::run` starts **one** listener: @@ -26,15 +29,20 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. | From | Event | To | live | ready | Notes | |---|---|---|---|---|---| | (unstarted) | `ProcessStarted` | `starting` | true | false | Construction. Live does not imply ready. | -| `starting` | `InitializationCompleted` | `loading_checkpoint` | true | false | `initialize()` succeeded; checkpoint still required. | +| `starting` | `InitializationCompleted` | `loading_checkpoint` | true | false | `initialize()` succeeded. Live daemon then applies the checkpoint stand-in (next row). | | `starting` | `InitializationFailed` | `fatal` | true | false | Sticky. Detail is JSON-only, never a metric label. | -| `loading_checkpoint` | `CheckpointValidated` | `running` | true | true | Ready only after this gate. | +| `loading_checkpoint` | `CheckpointValidated` | `running` | true | true | Ready only after this gate. Until LIM-1133 the daemon emits this right after `initialize`. | | `loading_checkpoint` | `CheckpointRejected` | `fatal` | true | false | Sticky. | | `running` | clock ≥ `stale_after` without ingress | `degraded` | true | true | Reason `stale_input`. Ready stays true. | | `degraded` (stale) | `IngressObserved` | `running` (if no other reasons) | true | true | Ticks without ingress do **not** clear stale. | | `running` | queue fill ≥ `overload_high` | `degraded` | true | true | Reason `overload`. | | `degraded` (overload) | fill ≤ `overload_low` | `running` (if no other reasons) | true | true | Hysteresis: mid-band does not recover. | | `running` / `degraded` | `BeginDrain` | `draining` | true | false | SIGTERM/SIGINT. Does not return to ready. | + +After `BeginDrain`, `BrainstemDaemon::run` stops the control listener. External +`/readyz` probes may get connection refused rather than `503`. In-process +`HealthHandle::snapshot()` still reports `phase: draining`. There is no probe +grace period. | any non-fatal | `Fatal` / init or checkpoint failure | `fatal` | true | false | Subsequent validate/tick/drain cannot restore ready. | Recoverable reasons (`stale_input`, `overload`) are independent: clearing one diff --git a/src/control.rs b/src/control.rs index 736ff56..4b5931b 100644 --- a/src/control.rs +++ b/src/control.rs @@ -21,6 +21,7 @@ use tracing::{info, warn}; use crate::health::{HealthHandle, HealthSnapshot}; const IO_TIMEOUT: Duration = Duration::from_secs(2); +const ACCEPT_BACKOFF: Duration = Duration::from_millis(50); const MAX_CONTROL_CONNS: usize = 32; pub async fn serve( @@ -57,7 +58,9 @@ pub async fn serve_listener( } } accepted = listener.accept() => { - spawn_accepted(accepted, &slots, &health); + if !spawn_accepted(accepted, &slots, &health) { + tokio::time::sleep(ACCEPT_BACKOFF).await; + } } } } @@ -80,10 +83,16 @@ fn spawn_accepted( accepted: std::io::Result<(TcpStream, std::net::SocketAddr)>, slots: &Arc, health: &HealthHandle, -) { +) -> bool { match accepted { - Ok((stream, _)) => spawn_control_conn(stream, slots, health), - Err(e) => warn!("control accept failed: {e}"), + Ok((stream, _)) => { + spawn_control_conn(stream, slots, health); + true + } + Err(e) => { + warn!("control accept failed: {e}"); + false + } } } diff --git a/src/health/handle.rs b/src/health/handle.rs index 12024c3..049c9b6 100644 --- a/src/health/handle.rs +++ b/src/health/handle.rs @@ -42,7 +42,11 @@ impl HealthHandle { /// Never waits. Returns `None` if a writer currently holds the lock. pub fn try_snapshot(&self) -> Option { - self.inner.try_read().ok().map(|guard| guard.snapshot()) + match self.inner.try_read() { + Ok(guard) => Some(guard.snapshot()), + Err(std::sync::TryLockError::Poisoned(e)) => Some(e.into_inner().snapshot()), + Err(std::sync::TryLockError::WouldBlock) => None, + } } #[cfg(test)] diff --git a/src/health/machine.rs b/src/health/machine.rs index 8a08c90..8a6f9d3 100644 --- a/src/health/machine.rs +++ b/src/health/machine.rs @@ -43,7 +43,11 @@ impl HealthLimits { overload_low: low, } } else { - Self::default() + Self { + stale_after: self.stale_after, + overload_high: 0.90, + overload_low: 0.70, + } } } } diff --git a/src/health/tests.rs b/src/health/tests.rs index 5d9f361..6eab965 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -447,3 +447,16 @@ fn non_finite_overload_limits_are_sanitized() { }); assert!(machine.snapshot().queue_pressure.overloaded); } + +#[test] +fn inverted_overload_limits_keep_stale_after() { + let limits = HealthLimits { + stale_after: Duration::from_millis(100), + overload_high: 0.50, + overload_low: 0.90, + } + .sanitized(); + assert_eq!(limits.stale_after, Duration::from_millis(100)); + assert_eq!(limits.overload_high, 0.90); + assert_eq!(limits.overload_low, 0.70); +} From 192119d6871d746db9cc08715bbb2e64ae08490f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:29:02 +0000 Subject: [PATCH 09/18] docs: qualify stub backend socket claim for control_bind The feature table said the stub binary opens no sockets. That is true for the backend, but control_bind still binds the optional HTTP listener. Co-authored-by: Raul Cardenas Montoya --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 30b0e98..b03ebdd 100644 --- a/README.md +++ b/README.md @@ -101,7 +101,7 @@ Default Cargo features are empty (`default = []` in `Cargo.toml`). That path use | Cargo flags | Wired backend | `libzmq` | Binary (`brainstem-daemon`) | Library `BrainstemDaemon::new()` / `try_new()` | |---|---|---|---|---| -| default / `--no-default-features` | stub | not required | no sockets; logs `🔌 Using stub backend` | stub | +| default / `--no-default-features` | stub | not required | no backend sockets by default; `control_bind` opens the control listener; logs `🔌 Using stub backend` | stub | | `--features corpus-ipc` | ZMQ / `corpus-ipc` | required | SUB via env, PUB on `spine_pub_port`; logs `📡 Using ZMQ corpus-ipc backend` | **still stub** | | `--all-features` | same as `corpus-ipc` | required | same as `--features corpus-ipc` | **still stub** | From 594ba75255e66dda6bb01ce59d3855236fa73857 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:30:55 +0000 Subject: [PATCH 10/18] fix: keep health docs table intact and tighten clock/tests Move the drain-probe note below the transition table, saturate FakeClock offsets so simulated time cannot wrap backward, and assert readiness from checkpoint validation alone. Co-authored-by: Raul Cardenas Montoya --- docs/health.md | 2 +- src/health/clock.rs | 6 +++++- src/health/tests.rs | 8 +++++--- 3 files changed, 11 insertions(+), 5 deletions(-) diff --git a/docs/health.md b/docs/health.md index 04edaa3..50a1a67 100644 --- a/docs/health.md +++ b/docs/health.md @@ -38,12 +38,12 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. | `running` | queue fill ≥ `overload_high` | `degraded` | true | true | Reason `overload`. | | `degraded` (overload) | fill ≤ `overload_low` | `running` (if no other reasons) | true | true | Hysteresis: mid-band does not recover. | | `running` / `degraded` | `BeginDrain` | `draining` | true | false | SIGTERM/SIGINT. Does not return to ready. | +| any non-fatal | `Fatal` / init or checkpoint failure | `fatal` | true | false | Subsequent validate/tick/drain cannot restore ready. | After `BeginDrain`, `BrainstemDaemon::run` stops the control listener. External `/readyz` probes may get connection refused rather than `503`. In-process `HealthHandle::snapshot()` still reports `phase: draining`. There is no probe grace period. -| any non-fatal | `Fatal` / init or checkpoint failure | `fatal` | true | false | Subsequent validate/tick/drain cannot restore ready. | Recoverable reasons (`stale_input`, `overload`) are independent: clearing one leaves the other. `capacity == 0` means “no queue instrumented” (LIM-1216) and diff --git a/src/health/clock.rs b/src/health/clock.rs index 03bf2a0..61a7c62 100644 --- a/src/health/clock.rs +++ b/src/health/clock.rs @@ -34,7 +34,11 @@ impl FakeClock { pub fn advance(&self, duration: Duration) { let add = u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX); - self.offset_nanos.fetch_add(add, Ordering::SeqCst); + let _ = self + .offset_nanos + .fetch_update(Ordering::SeqCst, Ordering::SeqCst, |current| { + Some(current.saturating_add(add)) + }); } } diff --git a/src/health/tests.rs b/src/health/tests.rs index 6eab965..608b22b 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -61,7 +61,9 @@ fn initialization_does_not_imply_readiness() { #[test] fn checkpoint_validation_makes_ready() { let (mut m, _) = machine(); - bring_ready(&mut m); + m.apply(HealthEvent::ProcessStarted); + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); let snap = m.snapshot(); assert!(snap.live); assert!(snap.ready); @@ -71,7 +73,7 @@ fn checkpoint_validation_makes_ready() { snap.checkpoint.as_ref().map(|c| c.id.as_str()), Some("soma16") ); - assert_eq!(snap.last_successful_tick_ms, Some(0)); + assert_eq!(snap.last_successful_tick_ms, None); assert!(!snap.input_freshness.stale); } @@ -125,7 +127,7 @@ fn stale_input_degrades_and_recovers_only_after_fresh_ingress() { } #[test] -fn overload_uses_hysteresis_and_clears_only_below_low_watermark() { +fn overload_uses_hysteresis_and_clears_at_or_below_low_watermark() { let (mut m, _) = machine(); bring_ready(&mut m); From e0e294ba5b995c51612c336e7bddb4c7598e468b Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:41:48 +0000 Subject: [PATCH 11/18] fix: reject equal overload watermarks and ignore pre-start init Equal high/low thresholds toggled overload on the same fill ratio. Initialization before ProcessStarted could report running while not live. Documented snapshot shapes now use digest null to match the LIM-1133 stand-in. Co-authored-by: Raul Cardenas Montoya --- src/health/machine.rs | 8 ++++---- src/health/tests.rs | 38 ++++++++++++++++++++++++++++++++++++-- 2 files changed, 40 insertions(+), 6 deletions(-) diff --git a/src/health/machine.rs b/src/health/machine.rs index 8a6f9d3..0dda7fa 100644 --- a/src/health/machine.rs +++ b/src/health/machine.rs @@ -32,11 +32,11 @@ impl Default for HealthLimits { } impl HealthLimits { - /// Replace non-finite or inverted watermarks with the built-in defaults. + /// Replace non-finite, inverted, or equal watermarks with the built-in defaults. pub fn sanitized(self) -> Self { let high = finite_or(self.overload_high, 0.90); let low = finite_or(self.overload_low, 0.70); - if high >= low { + if high > low { Self { stale_after: self.stale_after, overload_high: high, @@ -153,7 +153,7 @@ impl HealthMachine { } fn mark_initialized(&mut self) { - if self.fatal.is_none() && !self.draining { + if self.started && self.fatal.is_none() && !self.draining { self.initialized = true; } } @@ -175,7 +175,7 @@ impl HealthMachine { } fn can_accept_checkpoint(&self) -> bool { - self.fatal.is_none() && !self.draining && self.initialized + self.started && self.fatal.is_none() && !self.draining && self.initialized } fn apply_terminal(&mut self, event: HealthEvent) { diff --git a/src/health/tests.rs b/src/health/tests.rs index 608b22b..efe0b2c 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -58,6 +58,18 @@ fn initialization_does_not_imply_readiness() { assert_eq!(snap.reasons, vec![ReasonCode::CheckpointPending]); } +#[test] +fn initialization_before_start_is_ignored() { + let (mut m, _) = machine(); + m.apply(HealthEvent::InitializationCompleted); + m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + let snap = m.snapshot(); + assert!(!snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Starting); + assert_eq!(snap.reasons, vec![ReasonCode::Starting]); +} + #[test] fn checkpoint_validation_makes_ready() { let (mut m, _) = machine(); @@ -372,7 +384,16 @@ fn try_snapshot_does_not_block_on_write_lock() { #[test] fn example_snapshots_match_documented_shapes() { let (mut healthy, _) = machine(); - bring_ready(&mut healthy); + healthy.apply(HealthEvent::ProcessStarted); + healthy.apply(HealthEvent::InitializationCompleted); + healthy.apply(HealthEvent::CheckpointValidated { + identity: CheckpointIdentity { + id: "soma16".into(), + digest: None, + }, + }); + healthy.apply(HealthEvent::IngressObserved); + healthy.apply(HealthEvent::TickSucceeded); let healthy = healthy.snapshot(); assert_eq!( serde_json::to_value(&healthy).unwrap(), @@ -383,7 +404,7 @@ fn example_snapshots_match_documented_shapes() { "reasons": [], "last_successful_tick_ms": 0, "tick_age_ms": 0, - "checkpoint": { "id": "soma16", "digest": "abc123" }, + "checkpoint": { "id": "soma16", "digest": null }, "input_freshness": { "age_ms": 0, "stale": false }, "queue_pressure": { "depth": 0, "capacity": 0, "ratio": null, "overloaded": false }, "fatal": null, @@ -462,3 +483,16 @@ fn inverted_overload_limits_keep_stale_after() { assert_eq!(limits.overload_high, 0.90); assert_eq!(limits.overload_low, 0.70); } + +#[test] +fn equal_overload_watermarks_are_replaced() { + let limits = HealthLimits { + stale_after: Duration::from_millis(100), + overload_high: 0.80, + overload_low: 0.80, + } + .sanitized(); + assert_eq!(limits.stale_after, Duration::from_millis(100)); + assert_eq!(limits.overload_high, 0.90); + assert_eq!(limits.overload_low, 0.70); +} From 724e876d2742b087fbe5fbd5255c118e8d43ea66 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 06:45:03 +0000 Subject: [PATCH 12/18] fix: clamp overload watermarks to [0, 1] and drop flaky timing Out-of-range fill ratios could skip or never clear overload. try_snapshot coverage now asserts the non-blocking None result without a wall-clock bound. Co-authored-by: Raul Cardenas Montoya --- src/health/machine.rs | 4 ++-- src/health/tests.rs | 17 ++++++++++++++--- 2 files changed, 16 insertions(+), 5 deletions(-) diff --git a/src/health/machine.rs b/src/health/machine.rs index 0dda7fa..ef9eded 100644 --- a/src/health/machine.rs +++ b/src/health/machine.rs @@ -32,11 +32,11 @@ impl Default for HealthLimits { } impl HealthLimits { - /// Replace non-finite, inverted, or equal watermarks with the built-in defaults. + /// Replace non-finite, out-of-range, inverted, or equal watermarks with the built-in defaults. pub fn sanitized(self) -> Self { let high = finite_or(self.overload_high, 0.90); let low = finite_or(self.overload_low, 0.70); - if high > low { + if (0.0..=1.0).contains(&high) && (0.0..=1.0).contains(&low) && high > low { Self { stale_after: self.stale_after, overload_high: high, diff --git a/src/health/tests.rs b/src/health/tests.rs index efe0b2c..6e0df47 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -1,5 +1,5 @@ use super::*; -use std::time::{Duration, Instant}; +use std::time::Duration; fn limits() -> HealthLimits { HealthLimits { @@ -369,12 +369,10 @@ fn prometheus_labels_are_low_cardinality_and_omit_detail() { #[test] fn try_snapshot_does_not_block_on_write_lock() { let handle = HealthHandle::started(limits()); - let start = Instant::now(); { let _guard = handle.lock_write_for_test(); assert!(handle.try_snapshot().is_none()); } - assert!(start.elapsed() < Duration::from_millis(50)); assert!(handle.try_snapshot().is_some()); let snap = handle.snapshot(); assert!(snap.live); @@ -496,3 +494,16 @@ fn equal_overload_watermarks_are_replaced() { assert_eq!(limits.overload_high, 0.90); assert_eq!(limits.overload_low, 0.70); } + +#[test] +fn out_of_range_overload_watermarks_are_replaced() { + let limits = HealthLimits { + stale_after: Duration::from_millis(100), + overload_high: 1.5, + overload_low: -0.1, + } + .sanitized(); + assert_eq!(limits.stale_after, Duration::from_millis(100)); + assert_eq!(limits.overload_high, 0.90); + assert_eq!(limits.overload_low, 0.70); +} From 1a47c8e4b486dfe0b80f03544b7fa0b91856e912 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 07:01:33 +0000 Subject: [PATCH 13/18] test: split documented health snapshots under Codacy's 50-line limit Codacy rejected example_snapshots_match_documented_shapes (55 lines). Split the documented JSON, degraded, and fatal cases, and share the stand-in ready seeding helper so the digest-null path stays explicit. Co-authored-by: Raul Cardenas Montoya --- src/health/tests.rs | 40 +++++++++++++++++++++++++--------------- 1 file changed, 25 insertions(+), 15 deletions(-) diff --git a/src/health/tests.rs b/src/health/tests.rs index 6e0df47..dad3a84 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -22,14 +22,28 @@ fn ckpt() -> CheckpointIdentity { } } -fn bring_ready(m: &mut HealthMachine) { +fn seed_ready(m: &mut HealthMachine, identity: CheckpointIdentity) { m.apply(HealthEvent::ProcessStarted); m.apply(HealthEvent::InitializationCompleted); - m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::CheckpointValidated { identity }); m.apply(HealthEvent::IngressObserved); m.apply(HealthEvent::TickSucceeded); } +fn bring_ready(m: &mut HealthMachine) { + seed_ready(m, ckpt()); +} + +fn standin_ready(m: &mut HealthMachine) { + seed_ready( + m, + CheckpointIdentity { + id: "soma16".into(), + digest: None, + }, + ); +} + #[test] fn process_start_is_live_but_not_ready() { let (mut m, _) = machine(); @@ -380,21 +394,11 @@ fn try_snapshot_does_not_block_on_write_lock() { } #[test] -fn example_snapshots_match_documented_shapes() { +fn documented_running_snapshot_uses_null_digest() { let (mut healthy, _) = machine(); - healthy.apply(HealthEvent::ProcessStarted); - healthy.apply(HealthEvent::InitializationCompleted); - healthy.apply(HealthEvent::CheckpointValidated { - identity: CheckpointIdentity { - id: "soma16".into(), - digest: None, - }, - }); - healthy.apply(HealthEvent::IngressObserved); - healthy.apply(HealthEvent::TickSucceeded); - let healthy = healthy.snapshot(); + standin_ready(&mut healthy); assert_eq!( - serde_json::to_value(&healthy).unwrap(), + serde_json::to_value(healthy.snapshot()).unwrap(), serde_json::json!({ "live": true, "ready": true, @@ -409,7 +413,10 @@ fn example_snapshots_match_documented_shapes() { "observed_at_ms": 0 }) ); +} +#[test] +fn documented_degraded_snapshot_stays_ready() { let (mut degraded, clock) = machine(); bring_ready(&mut degraded); degraded.apply(HealthEvent::QueuePressure { @@ -424,7 +431,10 @@ fn example_snapshots_match_documented_shapes() { vec![ReasonCode::StaleInput, ReasonCode::Overload] ); assert!(degraded.ready); +} +#[test] +fn documented_fatal_snapshot_drops_readiness() { let (mut fatal, _) = machine(); fatal.apply(HealthEvent::ProcessStarted); fatal.apply(HealthEvent::InitializationCompleted); From ccb18f50baca25b2744e12fdfc68a8024a4a4ea6 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 07:13:10 +0000 Subject: [PATCH 14/18] fix: ignore pre-start failures and reply 503 when control is saturated InitializationFailed and CheckpointRejected now no-op until ProcessStarted, matching the existing out-of-order success-event guards. Saturated control accepts write a bounded 503 busy reply instead of dropping the stream. Co-authored-by: Raul Cardenas Montoya --- docs/health.md | 4 +++- src/control.rs | 43 ++++++++++++++++++++++++++++++++++++++----- src/health/machine.rs | 10 ++++++++-- src/health/tests.rs | 21 ++++++++++++++++++++- 4 files changed, 69 insertions(+), 9 deletions(-) diff --git a/docs/health.md b/docs/health.md index 50a1a67..89639b7 100644 --- a/docs/health.md +++ b/docs/health.md @@ -29,6 +29,7 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. | From | Event | To | live | ready | Notes | |---|---|---|---|---|---| | (unstarted) | `ProcessStarted` | `starting` | true | false | Construction. Live does not imply ready. | +| (unstarted) | init/checkpoint success or failure | (unstarted) | false | false | Ignored until `ProcessStarted`, matching out-of-order success handling. | | `starting` | `InitializationCompleted` | `loading_checkpoint` | true | false | `initialize()` succeeded. Live daemon then applies the checkpoint stand-in (next row). | | `starting` | `InitializationFailed` | `fatal` | true | false | Sticky. Detail is JSON-only, never a metric label. | | `loading_checkpoint` | `CheckpointValidated` | `running` | true | true | Ready only after this gate. Until LIM-1133 the daemon emits this right after `initialize`. | @@ -43,7 +44,8 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. After `BeginDrain`, `BrainstemDaemon::run` stops the control listener. External `/readyz` probes may get connection refused rather than `503`. In-process `HealthHandle::snapshot()` still reports `phase: draining`. There is no probe -grace period. +grace period. If all 32 in-flight control slots are busy, a new connection gets +a bounded `503` (`busy`) instead of a silent drop. Recoverable reasons (`stale_input`, `overload`) are independent: clearing one leaves the other. `capacity == 0` means “no queue instrumented” (LIM-1216) and diff --git a/src/control.rs b/src/control.rs index 4b5931b..b09b512 100644 --- a/src/control.rs +++ b/src/control.rs @@ -15,7 +15,7 @@ use std::time::Duration; use anyhow::{Context, Result}; use tokio::io::{AsyncReadExt, AsyncWriteExt}; use tokio::net::{TcpListener, TcpStream}; -use tokio::sync::{Semaphore, watch}; +use tokio::sync::{OwnedSemaphorePermit, Semaphore, watch}; use tracing::{info, warn}; use crate::health::{HealthHandle, HealthSnapshot}; @@ -97,10 +97,13 @@ fn spawn_accepted( } fn spawn_control_conn(stream: TcpStream, slots: &Arc, health: &HealthHandle) { - let Ok(permit) = slots.clone().try_acquire_owned() else { - return; - }; - let health = health.clone(); + match slots.clone().try_acquire_owned() { + Ok(permit) => spawn_served_conn(stream, permit, health.clone()), + Err(_) => spawn_busy_conn(stream), + } +} + +fn spawn_served_conn(stream: TcpStream, permit: OwnedSemaphorePermit, health: HealthHandle) { tokio::spawn(async move { let _permit = permit; if let Err(e) = handle_connection(stream, &health).await { @@ -109,6 +112,16 @@ fn spawn_control_conn(stream: TcpStream, slots: &Arc, health: &Health }); } +fn spawn_busy_conn(stream: TcpStream) { + tokio::spawn(async move { + let mut stream = stream; + let response = http_response(503, "text/plain; charset=utf-8", b"busy\n"); + if let Err(e) = write_http_response(&mut stream, &response).await { + warn!("control busy reply failed: {e}"); + } + }); +} + async fn handle_connection(mut stream: TcpStream, health: &HealthHandle) -> Result<()> { let req = read_http_request(&mut stream).await?; let response = render_http(&req, &health.snapshot()); @@ -335,6 +348,26 @@ mod tests { server.await.unwrap().unwrap(); } + #[tokio::test] + async fn saturated_control_writes_503() { + let slots = Arc::new(Semaphore::new(0)); + let health = HealthHandle::started(HealthLimits::default()); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let addr = listener.local_addr().unwrap(); + let mut client = TcpStream::connect(addr).await.unwrap(); + let (server, _) = listener.accept().await.unwrap(); + spawn_control_conn(server, &slots, &health); + let mut buf = vec![0u8; 512]; + let n = tokio::time::timeout(Duration::from_secs(2), client.read(&mut buf)) + .await + .unwrap() + .unwrap(); + let raw = String::from_utf8_lossy(&buf[..n]); + assert!(raw.contains("HTTP/1.1 503"), "{raw}"); + assert!(raw.contains("busy"), "{raw}"); + assert_eq!(slots.available_permits(), 0); + } + async fn http_get(addr: SocketAddr, path: &str) -> String { let mut stream = tokio::net::TcpStream::connect(addr).await.unwrap(); stream diff --git a/src/health/machine.rs b/src/health/machine.rs index ef9eded..f1c5d67 100644 --- a/src/health/machine.rs +++ b/src/health/machine.rs @@ -139,7 +139,7 @@ impl HealthMachine { HealthEvent::ProcessStarted => self.mark_started(now), HealthEvent::InitializationCompleted => self.mark_initialized(), HealthEvent::InitializationFailed { detail } => { - self.enter_fatal(FatalCode::InitializationFailed, detail); + self.fatal_if_started(FatalCode::InitializationFailed, detail); } _ => {} } @@ -168,7 +168,7 @@ impl HealthMachine { } } HealthEvent::CheckpointRejected { detail } => { - self.enter_fatal(FatalCode::CheckpointInvalid, detail); + self.fatal_if_started(FatalCode::CheckpointInvalid, detail); } _ => {} } @@ -333,6 +333,12 @@ impl HealthMachine { } } + fn fatal_if_started(&mut self, code: FatalCode, detail: String) { + if self.started { + self.enter_fatal(code, detail); + } + } + fn enter_fatal(&mut self, code: FatalCode, detail: String) { if self.fatal.is_some() { return; diff --git a/src/health/tests.rs b/src/health/tests.rs index dad3a84..0444517 100644 --- a/src/health/tests.rs +++ b/src/health/tests.rs @@ -77,13 +77,32 @@ fn initialization_before_start_is_ignored() { let (mut m, _) = machine(); m.apply(HealthEvent::InitializationCompleted); m.apply(HealthEvent::CheckpointValidated { identity: ckpt() }); + m.apply(HealthEvent::ProcessStarted); let snap = m.snapshot(); - assert!(!snap.live); + assert!(snap.live); assert!(!snap.ready); assert_eq!(snap.phase, HealthPhase::Starting); + assert!(snap.checkpoint.is_none()); assert_eq!(snap.reasons, vec![ReasonCode::Starting]); } +#[test] +fn initialization_failure_before_start_is_ignored() { + let (mut m, _) = machine(); + m.apply(HealthEvent::InitializationFailed { + detail: "too early".into(), + }); + m.apply(HealthEvent::CheckpointRejected { + detail: "too early".into(), + }); + m.apply(HealthEvent::ProcessStarted); + let snap = m.snapshot(); + assert!(snap.live); + assert!(!snap.ready); + assert_eq!(snap.phase, HealthPhase::Starting); + assert!(snap.fatal.is_none()); +} + #[test] fn checkpoint_validation_makes_ready() { let (mut m, _) = machine(); From 9893bbbc4cd339f591f9b8ef99acf278d2d8f845 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 07:22:18 +0000 Subject: [PATCH 15/18] fix: scope fatal docs to started states and use try_snapshot on control The transition table no longer claims unstarted init/checkpoint failures become fatal. Control requests take try_snapshot and return 503 busy when the write lock is held, so probes stay bounded. Co-authored-by: Raul Cardenas Montoya --- docs/health.md | 7 ++++--- src/control.rs | 27 +++++++++++++++++++++++---- 2 files changed, 27 insertions(+), 7 deletions(-) diff --git a/docs/health.md b/docs/health.md index 89639b7..47c13e0 100644 --- a/docs/health.md +++ b/docs/health.md @@ -39,13 +39,14 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. | `running` | queue fill ≥ `overload_high` | `degraded` | true | true | Reason `overload`. | | `degraded` (overload) | fill ≤ `overload_low` | `running` (if no other reasons) | true | true | Hysteresis: mid-band does not recover. | | `running` / `degraded` | `BeginDrain` | `draining` | true | false | SIGTERM/SIGINT. Does not return to ready. | -| any non-fatal | `Fatal` / init or checkpoint failure | `fatal` | true | false | Subsequent validate/tick/drain cannot restore ready. | +| any started non-fatal | `Fatal` / init or checkpoint failure | `fatal` | true | false | Subsequent validate/tick/drain cannot restore ready. Unstarted init/checkpoint failures stay ignored. | After `BeginDrain`, `BrainstemDaemon::run` stops the control listener. External `/readyz` probes may get connection refused rather than `503`. In-process `HealthHandle::snapshot()` still reports `phase: draining`. There is no probe -grace period. If all 32 in-flight control slots are busy, a new connection gets -a bounded `503` (`busy`) instead of a silent drop. +grace period. If all 32 in-flight control slots are busy, or a snapshot read +would block on an in-flight `apply`, the connection gets a bounded `503` +(`busy`) instead of waiting or dropping. Recoverable reasons (`stale_input`, `overload`) are independent: clearing one leaves the other. `capacity == 0` means “no queue instrumented” (LIM-1216) and diff --git a/src/control.rs b/src/control.rs index b09b512..8c943f9 100644 --- a/src/control.rs +++ b/src/control.rs @@ -115,8 +115,7 @@ fn spawn_served_conn(stream: TcpStream, permit: OwnedSemaphorePermit, health: He fn spawn_busy_conn(stream: TcpStream) { tokio::spawn(async move { let mut stream = stream; - let response = http_response(503, "text/plain; charset=utf-8", b"busy\n"); - if let Err(e) = write_http_response(&mut stream, &response).await { + if let Err(e) = write_http_response(&mut stream, &busy_http()).await { warn!("control busy reply failed: {e}"); } }); @@ -124,8 +123,18 @@ fn spawn_busy_conn(stream: TcpStream) { async fn handle_connection(mut stream: TcpStream, health: &HealthHandle) -> Result<()> { let req = read_http_request(&mut stream).await?; - let response = render_http(&req, &health.snapshot()); - write_http_response(&mut stream, &response).await + write_http_response(&mut stream, &control_response(&req, health)).await +} + +fn control_response(req: &str, health: &HealthHandle) -> Vec { + match health.try_snapshot() { + Some(snapshot) => render_http(req, &snapshot), + None => busy_http(), + } +} + +fn busy_http() -> Vec { + http_response(503, "text/plain; charset=utf-8", b"busy\n") } async fn read_http_request(stream: &mut TcpStream) -> Result { @@ -368,6 +377,16 @@ mod tests { assert_eq!(slots.available_permits(), 0); } + #[test] + fn control_response_is_503_when_snapshot_would_block() { + let health = HealthHandle::started(HealthLimits::default()); + let _guard = health.lock_write_for_test(); + let raw = + String::from_utf8(control_response("GET /livez HTTP/1.1\r\n\r\n", &health)).unwrap(); + assert!(raw.starts_with("HTTP/1.1 503")); + assert!(raw.contains("busy")); + } + async fn http_get(addr: SocketAddr, path: &str) -> String { let mut stream = tokio::net::TcpStream::connect(addr).await.unwrap(); stream From 0cafdea381b8168a9336b90a1ae5ec7cbe268f8f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 07:27:52 +0000 Subject: [PATCH 16/18] fix: bound busy control replies and clarify monotonic clock Saturated accepts now take one of four short 503 slots; further connections are closed immediately so busy writes cannot grow without bound. SystemClock is documented as monotonic, and the unstarted fatal row points at ProcessStarted. Co-authored-by: Raul Cardenas Montoya --- docs/health.md | 8 +++---- src/control.rs | 57 ++++++++++++++++++++++++++++++++++++++------ src/health/clock.rs | 2 +- src/health/handle.rs | 5 +++- 4 files changed, 59 insertions(+), 13 deletions(-) diff --git a/docs/health.md b/docs/health.md index 47c13e0..ff2f89c 100644 --- a/docs/health.md +++ b/docs/health.md @@ -29,7 +29,7 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. | From | Event | To | live | ready | Notes | |---|---|---|---|---|---| | (unstarted) | `ProcessStarted` | `starting` | true | false | Construction. Live does not imply ready. | -| (unstarted) | init/checkpoint success or failure | (unstarted) | false | false | Ignored until `ProcessStarted`, matching out-of-order success handling. | +| (unstarted) | init/checkpoint success or failure | (unstarted) | false | false | Ignored until `ProcessStarted`; the fatal transition below applies only after `ProcessStarted`. | | `starting` | `InitializationCompleted` | `loading_checkpoint` | true | false | `initialize()` succeeded. Live daemon then applies the checkpoint stand-in (next row). | | `starting` | `InitializationFailed` | `fatal` | true | false | Sticky. Detail is JSON-only, never a metric label. | | `loading_checkpoint` | `CheckpointValidated` | `running` | true | true | Ready only after this gate. Until LIM-1133 the daemon emits this right after `initialize`. | @@ -44,9 +44,9 @@ waiting on the tick loop's backend or `SpikingNetwork::step`. After `BeginDrain`, `BrainstemDaemon::run` stops the control listener. External `/readyz` probes may get connection refused rather than `503`. In-process `HealthHandle::snapshot()` still reports `phase: draining`. There is no probe -grace period. If all 32 in-flight control slots are busy, or a snapshot read -would block on an in-flight `apply`, the connection gets a bounded `503` -(`busy`) instead of waiting or dropping. +grace period. If all 32 in-flight control slots are busy, up to 4 extra +connections get a short `503` (`busy`); further accepts are closed immediately. +The same `503` is used when a snapshot read would block on an in-flight `apply`. Recoverable reasons (`stale_input`, `overload`) are independent: clearing one leaves the other. `capacity == 0` means “no queue instrumented” (LIM-1216) and diff --git a/src/control.rs b/src/control.rs index 8c943f9..f51f9c3 100644 --- a/src/control.rs +++ b/src/control.rs @@ -21,8 +21,10 @@ use tracing::{info, warn}; use crate::health::{HealthHandle, HealthSnapshot}; const IO_TIMEOUT: Duration = Duration::from_secs(2); +const BUSY_IO_TIMEOUT: Duration = Duration::from_millis(50); const ACCEPT_BACKOFF: Duration = Duration::from_millis(50); const MAX_CONTROL_CONNS: usize = 32; +const MAX_BUSY_CONNS: usize = 4; pub async fn serve( addr: SocketAddr, @@ -50,6 +52,7 @@ pub async fn serve_listener( } let slots = Arc::new(Semaphore::new(MAX_CONTROL_CONNS)); + let busy_slots = Arc::new(Semaphore::new(MAX_BUSY_CONNS)); loop { tokio::select! { changed = shutdown.changed() => { @@ -58,7 +61,7 @@ pub async fn serve_listener( } } accepted = listener.accept() => { - if !spawn_accepted(accepted, &slots, &health) { + if !spawn_accepted(accepted, &slots, &busy_slots, &health) { tokio::time::sleep(ACCEPT_BACKOFF).await; } } @@ -82,11 +85,12 @@ fn should_stop_control( fn spawn_accepted( accepted: std::io::Result<(TcpStream, std::net::SocketAddr)>, slots: &Arc, + busy_slots: &Arc, health: &HealthHandle, ) -> bool { match accepted { Ok((stream, _)) => { - spawn_control_conn(stream, slots, health); + spawn_control_conn(stream, slots, busy_slots, health); true } Err(e) => { @@ -96,10 +100,15 @@ fn spawn_accepted( } } -fn spawn_control_conn(stream: TcpStream, slots: &Arc, health: &HealthHandle) { +fn spawn_control_conn( + stream: TcpStream, + slots: &Arc, + busy_slots: &Arc, + health: &HealthHandle, +) { match slots.clone().try_acquire_owned() { Ok(permit) => spawn_served_conn(stream, permit, health.clone()), - Err(_) => spawn_busy_conn(stream), + Err(_) => spawn_busy_conn(stream, busy_slots), } } @@ -112,10 +121,14 @@ fn spawn_served_conn(stream: TcpStream, permit: OwnedSemaphorePermit, health: He }); } -fn spawn_busy_conn(stream: TcpStream) { +fn spawn_busy_conn(stream: TcpStream, busy_slots: &Arc) { + let Ok(permit) = busy_slots.clone().try_acquire_owned() else { + return; + }; tokio::spawn(async move { + let _permit = permit; let mut stream = stream; - if let Err(e) = write_http_response(&mut stream, &busy_http()).await { + if let Err(e) = write_busy_http(&mut stream).await { warn!("control busy reply failed: {e}"); } }); @@ -137,6 +150,17 @@ fn busy_http() -> Vec { http_response(503, "text/plain; charset=utf-8", b"busy\n") } +async fn write_busy_http(stream: &mut TcpStream) -> Result<()> { + tokio::time::timeout(BUSY_IO_TIMEOUT, async { + stream.write_all(&busy_http()).await?; + stream.flush().await?; + Ok::<(), std::io::Error>(()) + }) + .await + .context("control busy write timed out")? + .context("control busy write failed") +} + async fn read_http_request(stream: &mut TcpStream) -> Result { let mut buf = [0u8; 1024]; let n = timed_io( @@ -365,7 +389,8 @@ mod tests { let addr = listener.local_addr().unwrap(); let mut client = TcpStream::connect(addr).await.unwrap(); let (server, _) = listener.accept().await.unwrap(); - spawn_control_conn(server, &slots, &health); + let busy_slots = Arc::new(Semaphore::new(MAX_BUSY_CONNS)); + spawn_control_conn(server, &slots, &busy_slots, &health); let mut buf = vec![0u8; 512]; let n = tokio::time::timeout(Duration::from_secs(2), client.read(&mut buf)) .await @@ -377,6 +402,24 @@ mod tests { assert_eq!(slots.available_permits(), 0); } + #[tokio::test] + async fn fully_saturated_control_closes_without_reply() { + let slots = Arc::new(Semaphore::new(0)); + let busy_slots = Arc::new(Semaphore::new(0)); + let health = HealthHandle::started(HealthLimits::default()); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let addr = listener.local_addr().unwrap(); + let mut client = TcpStream::connect(addr).await.unwrap(); + let (server, _) = listener.accept().await.unwrap(); + spawn_control_conn(server, &slots, &busy_slots, &health); + let mut buf = vec![0u8; 512]; + let result = tokio::time::timeout(Duration::from_millis(200), client.read(&mut buf)).await; + assert!( + matches!(result, Ok(Ok(0)) | Ok(Err(_))), + "expected close, got {result:?}" + ); + } + #[test] fn control_response_is_503_when_snapshot_would_block() { let health = HealthHandle::started(HealthLimits::default()); diff --git a/src/health/clock.rs b/src/health/clock.rs index 61a7c62..3aab2eb 100644 --- a/src/health/clock.rs +++ b/src/health/clock.rs @@ -10,7 +10,7 @@ pub trait Clock: Send + Sync { fn now(&self) -> Instant; } -/// Wall-clock monotonic clock. +/// System monotonic clock. #[derive(Debug, Clone, Copy, Default)] pub struct SystemClock; diff --git a/src/health/handle.rs b/src/health/handle.rs index 049c9b6..bbaca13 100644 --- a/src/health/handle.rs +++ b/src/health/handle.rs @@ -7,7 +7,10 @@ use super::clock::SystemClock; use super::machine::{HealthEvent, HealthLimits, HealthMachine}; use super::snapshot::HealthSnapshot; -/// Cloneable, non-tick-blocking handle for supervisors and the control surface. +/// Cloneable handle for supervisors and the control surface. +/// +/// `try_snapshot` never waits. `snapshot` may wait only for an in-flight `apply` +/// (no I/O, no tick-loop backend). #[derive(Clone)] pub struct HealthHandle { inner: Arc>, From f87ff32dc49f600bad3de77b168a4aa42ea46a0f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 07:32:24 +0000 Subject: [PATCH 17/18] docs: qualify /health and /metrics 200 with 503 busy The path table claimed /health is always 200, which conflicts with the bounded busy reply when slots are full or try_snapshot would block. Co-authored-by: Raul Cardenas Montoya --- docs/health.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/health.md b/docs/health.md index ff2f89c..8cf6a3f 100644 --- a/docs/health.md +++ b/docs/health.md @@ -14,8 +14,8 @@ This repository had no HTTP/metrics server before this surface. When |---|---| | `GET /livez` | `200` if live, `503` otherwise | | `GET /readyz` | `200` if ready, `503` otherwise | -| `GET /health` | `200` JSON [`HealthSnapshot`](../src/health/snapshot.rs) (always; inspect `phase`) | -| `GET /metrics` | Prometheus text; labels are phase/reason codes only | +| `GET /health` | `200` JSON [`HealthSnapshot`](../src/health/snapshot.rs) (inspect `phase`); bounded `503 busy` if the control plane cannot snapshot | +| `GET /metrics` | Prometheus text; labels are phase/reason codes only; same `503 busy` exception | Leave `control_bind` unset to preserve the historical no-extra-socket default. Do not add a second control server beside this one. From f582e1dbda9be2001423d7607bb19eb942dc0a57 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Tue, 15 Sep 2026 07:40:36 +0000 Subject: [PATCH 18/18] docs: require Clock::now to be non-blocking Snapshots call now() while the handle read guard is held, so a custom clock must not block or perform I/O. Co-authored-by: Raul Cardenas Montoya --- src/health/clock.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/src/health/clock.rs b/src/health/clock.rs index 3aab2eb..e025dd7 100644 --- a/src/health/clock.rs +++ b/src/health/clock.rs @@ -6,7 +6,11 @@ use std::sync::atomic::{AtomicU64, Ordering}; use std::time::{Duration, Instant}; /// Monotonic clock used by the health state machine. +/// +/// [`Clock::now`] must return without blocking or I/O. Snapshots call it while a +/// handle read guard is held. pub trait Clock: Send + Sync { + /// Current monotonic time. Must not block or perform I/O. fn now(&self) -> Instant; }