From f5c2bbcae9a60684751c38cecd64c90bd6b0a270 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 16:25:27 -0700 Subject: [PATCH 01/18] fix(sandbox): accept local connections natively on loopback-confined sockets On kernels without SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV (RHEL 9 / RHCOS 5.14), the broker cannot safely write an accepted peer address into workload memory, so accept/accept4 with a peer-address buffer failed with EOPNOTSUPP. Static binaries and Go servers, which issue the raw syscall, could not accept connections at all. Move local acceptance into the kernel and replace per-accept inspection with standing kernel confinement: - Bind every broker-created TCP/UDP socket to the loopback device before injection and verify the binding. Accepted sockets inherit it, so they can neither receive routed ingress nor emit routed egress. - Stop notifying accept/accept4 and remove the accept workers, the SIGUSR2 accept-interrupt monitor, and the 64-accept ceiling. - Continue getpeername natively for every descriptor except relayed connections. - Deny interface-selection socket options and MSG_FASTOPEN sends from scalar syscall arguments, so confinement does not depend on the capability state of the user namespace that owns the network namespace and covers unregistered descriptors. - Drop loopback-interface ingress on the non-loopback TCP control listener before it listens, so a workload cannot reach it by reconnecting a natively accepted socket. - Mark descriptors above stdio close-on-exec in every workload pre_exec path. - Require an active confinement probe at qualification and carry it as required authenticated audit evidence. Document OpenShift 4.19 as the minimum release: RHCOS kernels for 4.16 through 4.18 are built without Landlock. Closes #4058 Signed-off-by: Drew Newberry --- Cargo.lock | 1 + .../openshell-isolation-interface/Cargo.toml | 1 + .../src/linux/child_seccomp.rs | 61 ++ .../src/linux/mod.rs | 1 + .../src/linux/seccomp_notify.rs | 6 +- .../src/linux/socket_confinement.rs | 540 +++++++++++++ .../src/linux/socket_registry.rs | 7 +- .../src/boundary_protocol.rs | 29 +- .../openshell-sandbox-backend/src/runtime.rs | 1 + .../openshell-sandbox/src/accept_interrupt.rs | 297 ------- .../openshell-sandbox/src/boundary_server.rs | 60 +- crates/openshell-sandbox/src/lib.rs | 3 +- crates/openshell-sandbox/src/main.rs | 8 + .../openshell-sandbox/src/network_broker.rs | 727 ++++++++++-------- docs/about/support-matrix.mdx | 15 +- docs/kubernetes/openshift.mdx | 13 +- 16 files changed, 1134 insertions(+), 636 deletions(-) create mode 100644 crates/openshell-isolation-interface/src/linux/socket_confinement.rs delete mode 100644 crates/openshell-sandbox/src/accept_interrupt.rs diff --git a/Cargo.lock b/Cargo.lock index fc0b173702..3fbee07a53 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4413,6 +4413,7 @@ dependencies = [ "serde", "serde_json", "sha2 0.10.9", + "socket2", "tokio", ] diff --git a/crates/openshell-isolation-interface/Cargo.toml b/crates/openshell-isolation-interface/Cargo.toml index 0426a9011f..b863f312a4 100644 --- a/crates/openshell-isolation-interface/Cargo.toml +++ b/crates/openshell-isolation-interface/Cargo.toml @@ -25,6 +25,7 @@ libc = "0.2" rustix = { workspace = true, features = ["fs", "process"] } [dev-dependencies] +socket2 = { workspace = true } tokio = { workspace = true } [lints] diff --git a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs index 4ba4eefafb..e709e755b7 100644 --- a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs +++ b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs @@ -29,6 +29,7 @@ const SECCOMP_DATA_ARGS_OFFSET: u32 = 16; const X32_SYSCALL_BIT: u32 = 0x4000_0000; const CLOSE_RANGE_UNSHARE_FLAG: u32 = 1 << 1; +const CLOSE_RANGE_CLOEXEC_FLAG: u32 = 1 << 2; const F_SETOWN_COMMAND: u32 = 8; const F_SETSIG_COMMAND: u32 = 10; const F_SETOWN_EX_COMMAND: u32 = 15; @@ -53,6 +54,7 @@ impl ChildHardeningProgram { /// The caller must invoke this from the post-fork child after all /// sandbox-wide TSYNC work and the launcher's `NEW_LISTENER` filter. pub fn install(&mut self) -> io::Result<()> { + mark_inherited_descriptors_close_on_exec()?; set_no_new_privileges()?; let len = u16::try_from(self.instructions.len()).map_err(|_| { io::Error::new( @@ -88,6 +90,35 @@ impl ChildHardeningProgram { } } +/// Mark every descriptor above stdio close-on-exec in the post-fork child. +/// +/// Workloads receive INET sockets only through broker injection, which binds +/// them to loopback first. A descriptor the sandbox process inherited from its +/// container runtime, or opened without `O_CLOEXEC`, must never cross `exec` +/// as an unconfined socket. The command's stdio is already installed on 0-2 +/// when `pre_exec` hooks run. This is a single async-signal-safe syscall. +/// +/// # Errors +/// +/// Returns the kernel error; kernels without `CLOSE_RANGE_CLOEXEC` (before +/// Linux 5.11) fail closed. +pub fn mark_inherited_descriptors_close_on_exec() -> io::Result<()> { + // SAFETY: close_range takes scalar arguments and only sets FD_CLOEXEC. + let result = unsafe { + libc::syscall( + libc::SYS_close_range, + 3_u32, + u32::MAX, + CLOSE_RANGE_CLOEXEC_FLAG, + ) + }; + if result < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(()) + } +} + /// Build the same-UID workload self-protection program before `fork`. /// /// `sandbox_tgid` is the sandbox PID as visible from its workload namespace. @@ -352,6 +383,36 @@ fn set_no_new_privileges() -> io::Result<()> { mod tests { use super::*; + #[test] + #[allow(unsafe_code)] + fn inherited_sockets_are_marked_close_on_exec_but_stdio_is_not() { + // SAFETY: scalar socket arguments; deliberately inheritable. + let socket = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0) }; + assert!(socket > 2); + // SAFETY: the child performs only async-signal-safe syscalls and exits. + let pid = unsafe { libc::fork() }; + assert!(pid >= 0); + if pid == 0 { + let swept = mark_inherited_descriptors_close_on_exec().is_ok(); + // SAFETY: F_GETFD reads one descriptor flag word. + let socket_flags = unsafe { libc::fcntl(socket, libc::F_GETFD) }; + // SAFETY: as above, for stderr. + let stderr_flags = unsafe { libc::fcntl(libc::STDERR_FILENO, libc::F_GETFD) }; + let ok = swept + && socket_flags & libc::FD_CLOEXEC != 0 + && stderr_flags >= 0 + && stderr_flags & libc::FD_CLOEXEC == 0; + // SAFETY: terminate the forked child without running destructors. + unsafe { libc::_exit(i32::from(!ok)) }; + } + let mut status = 0; + // SAFETY: wait for the child created above. + assert_eq!(unsafe { libc::waitpid(pid, &raw mut status, 0) }, pid); + // SAFETY: close the test-owned socket. + unsafe { libc::close(socket) }; + assert!(libc::WIFEXITED(status) && libc::WEXITSTATUS(status) == 0); + } + #[test] fn rejects_zero_sandbox_tgid() { assert_eq!( diff --git a/crates/openshell-isolation-interface/src/linux/mod.rs b/crates/openshell-isolation-interface/src/linux/mod.rs index bad3d329dc..a15b8b6f43 100644 --- a/crates/openshell-isolation-interface/src/linux/mod.rs +++ b/crates/openshell-isolation-interface/src/linux/mod.rs @@ -11,6 +11,7 @@ pub mod landlock; pub mod proc_fd; pub mod process_signal; pub mod seccomp_notify; +pub mod socket_confinement; pub mod socket_registry; pub mod task_memory; pub mod workload_launcher; diff --git a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs index a87dc080d5..fbd8b78f9c 100644 --- a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs +++ b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs @@ -430,7 +430,9 @@ pub fn install_listener(syscalls: &[i64]) -> io::Result { /// /// The filter mediates every syscall that can create, select, or materially /// reconfigure an INET endpoint. Connected `send()`/null-destination -/// `sendto()` retains the audited cBPF fast path. +/// `sendto()` retains the audited cBPF fast path. `accept`/`accept4` run +/// natively: the broker binds every INET socket to loopback before injecting +/// it, and accepted sockets inherit their listener's binding. pub fn install_workload_listener() -> io::Result { #[allow(unused_mut)] // SYS_open is unavailable on some architectures. let mut syscalls = vec![ @@ -438,8 +440,6 @@ pub fn install_workload_listener() -> io::Result { libc::SYS_connect, libc::SYS_bind, libc::SYS_listen, - libc::SYS_accept, - libc::SYS_accept4, libc::SYS_sendto, libc::SYS_sendmsg, libc::SYS_sendmmsg, diff --git a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs new file mode 100644 index 0000000000..faa8a5ba37 --- /dev/null +++ b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs @@ -0,0 +1,540 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Standing kernel confinement for workload INET sockets. +//! +//! Every workload INET socket is bound to the loopback device before its +//! descriptor is injected. The binding is kernel state on the socket itself, +//! so it survives `dup`, `fork`, `exec`, `AF_UNSPEC` disconnect, and is +//! inherited by sockets accepted from a confined listener. It restricts both +//! directions: route lookups are pinned to `lo`, and listener/UDP lookup only +//! matches packets that arrive on `lo`. Clearing or changing an existing +//! binding requires `CAP_NET_RAW` in the network namespace's owning user +//! namespace, which the capability-free sandbox and workload do not hold. + +#![allow(unsafe_code)] + +use std::ffi::CStr; +use std::io; +use std::mem::size_of; +use std::os::fd::{AsRawFd as _, FromRawFd as _, OwnedFd, RawFd}; + +const LOOPBACK_DEVICE: &CStr = c"lo"; + +/// Bind `fd` to the loopback device and verify the kernel recorded it. +/// +/// # Errors +/// +/// Returns the kernel error when the binding cannot be installed, or `EPERM` +/// when the socket is already bound to another device. +pub fn confine_to_loopback(fd: RawFd) -> io::Result<()> { + let name = LOOPBACK_DEVICE.to_bytes_with_nul(); + // SAFETY: `name` is a live NUL-terminated buffer for the duration of the call. + let result = unsafe { + libc::setsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_BINDTODEVICE, + name.as_ptr().cast(), + socklen(name.len())?, + ) + }; + if result < 0 { + return Err(io::Error::last_os_error()); + } + if bound_device(fd)?.as_deref() == Some(LOOPBACK_DEVICE.to_bytes()) { + Ok(()) + } else { + Err(io::Error::from_raw_os_error(libc::EPERM)) + } +} + +/// Return the device name `fd` is bound to, or `None` when unbound. +/// +/// # Errors +/// +/// Returns the kernel error from `getsockopt(SO_BINDTODEVICE)`. +pub fn bound_device(fd: RawFd) -> io::Result>> { + let mut name = [0_u8; libc::IFNAMSIZ]; + let mut length = socklen(name.len())?; + // SAFETY: `name` and `length` are live, writable outputs sized together. + let result = unsafe { + libc::getsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_BINDTODEVICE, + name.as_mut_ptr().cast(), + &raw mut length, + ) + }; + if result < 0 { + return Err(io::Error::last_os_error()); + } + let length = usize::try_from(length).unwrap_or(0).min(name.len()); + let name = name[..length] + .split(|byte| *byte == 0) + .next() + .unwrap_or_default(); + Ok((!name.is_empty()).then(|| name.to_vec())) +} + +/// Drop TCP/UDP ingress that arrives on the loopback interface. +/// +/// Attach this to a trusted listener whose legitimate clients are never in the +/// same network namespace. Matching the ingress interface rather than the +/// source address also rejects connections to the host's own non-loopback +/// address, which the kernel delivers through loopback. The filter is locked +/// so later code cannot remove it accidentally. +/// +/// # Errors +/// +/// Returns the kernel error when the filter cannot be attached or locked. +pub fn reject_loopback_ingress(fd: RawFd) -> io::Result<()> { + // SAFETY: LOOPBACK_DEVICE is a valid NUL-terminated interface name. + let index = unsafe { libc::if_nametoindex(LOOPBACK_DEVICE.as_ptr()) }; + if index == 0 { + return Err(io::Error::last_os_error()); + } + reject_ingress_interface(fd, index) +} + +fn reject_ingress_interface(fd: RawFd, index: u32) -> io::Result<()> { + // Ancillary loads use the documented negative offset encoding. + let ifindex_offset = (libc::SKF_AD_OFF + libc::SKF_AD_IFINDEX).cast_unsigned(); + let mut program = [ + filter_stmt(libc::BPF_LD | libc::BPF_W | libc::BPF_ABS, ifindex_offset), + filter_jump(libc::BPF_JMP | libc::BPF_JEQ | libc::BPF_K, index, 0, 1), + filter_stmt(libc::BPF_RET | libc::BPF_K, 0), + filter_stmt(libc::BPF_RET | libc::BPF_K, u32::MAX), + ]; + let filter = libc::sock_fprog { + len: u16::try_from(program.len()).map_err(io::Error::other)?, + filter: program.as_mut_ptr(), + }; + // SAFETY: `filter` references `program`, which outlives the call. + let result = unsafe { + libc::setsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_ATTACH_FILTER, + (&raw const filter).cast(), + socklen(size_of::())?, + ) + }; + if result < 0 { + return Err(io::Error::last_os_error()); + } + set_int_option(fd, libc::SOL_SOCKET, libc::SO_LOCK_FILTER, 1) +} + +/// Actively prove loopback confinement under the current runtime profile. +/// +/// For each supported workload socket type this installs the binding, proves +/// that the sandbox credentials cannot clear or replace it, and proves that a +/// stream accepted from a confined listener inherits it. IPv6 is skipped only +/// when the kernel does not provide the address family. +/// +/// # Errors +/// +/// Returns an error describing the first failed property. +pub fn probe_loopback_confinement() -> io::Result<()> { + for (domain, kind) in [ + (libc::AF_INET, libc::SOCK_STREAM), + (libc::AF_INET, libc::SOCK_DGRAM), + (libc::AF_INET6, libc::SOCK_STREAM), + (libc::AF_INET6, libc::SOCK_DGRAM), + ] { + let socket = match new_socket(domain, kind) { + Ok(socket) => socket, + Err(error) + if domain == libc::AF_INET6 && error.raw_os_error() == Some(libc::EAFNOSUPPORT) => + { + continue; + } + Err(error) => return Err(error), + }; + confine_to_loopback(socket.as_raw_fd()) + .map_err(|error| probe_error("install loopback binding", &error))?; + probe_binding_is_immutable(socket.as_raw_fd())?; + } + probe_accept_inherits_binding() +} + +fn probe_binding_is_immutable(fd: RawFd) -> io::Result<()> { + let empty = [0_u8; 1]; + // SAFETY: `empty` is a live one-byte buffer; an empty name requests unbind. + let clear = unsafe { + libc::setsockopt( + fd, + libc::SOL_SOCKET, + libc::SO_BINDTODEVICE, + empty.as_ptr().cast(), + socklen(empty.len())?, + ) + }; + if clear == 0 { + return Err(io::Error::other( + "sandbox credentials can clear a socket device binding", + )); + } + if set_int_option(fd, libc::SOL_SOCKET, libc::SO_BINDTOIFINDEX, 0).is_ok() { + return Err(io::Error::other( + "sandbox credentials can clear a socket interface-index binding", + )); + } + if bound_device(fd)?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + return Err(io::Error::other("socket device binding changed")); + } + Ok(()) +} + +fn probe_accept_inherits_binding() -> io::Result<()> { + let listener = std::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0))?; + confine_to_loopback(listener.as_raw_fd()) + .map_err(|error| probe_error("confine probe listener", &error))?; + let client = std::net::TcpStream::connect(listener.local_addr()?)?; + let (accepted, peer) = listener.accept()?; + if peer != client.local_addr()? { + return Err(io::Error::other("accepted probe peer mismatch")); + } + if bound_device(accepted.as_raw_fd())?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + return Err(io::Error::other( + "accepted socket did not inherit the loopback binding", + )); + } + Ok(()) +} + +fn probe_error(context: &str, error: &io::Error) -> io::Error { + io::Error::new(error.kind(), format!("{context}: {error}")) +} + +fn new_socket(domain: i32, kind: i32) -> io::Result { + // SAFETY: scalar socket arguments; success returns one owned descriptor. + let fd = unsafe { libc::socket(domain, kind | libc::SOCK_CLOEXEC, 0) }; + if fd < 0 { + return Err(io::Error::last_os_error()); + } + // SAFETY: successful socket returned one newly owned descriptor. + Ok(unsafe { OwnedFd::from_raw_fd(fd) }) +} + +fn set_int_option(fd: RawFd, level: i32, option: i32, value: i32) -> io::Result<()> { + // SAFETY: `value` is a live int for the duration of the call. + let result = unsafe { + libc::setsockopt( + fd, + level, + option, + (&raw const value).cast(), + socklen(size_of::())?, + ) + }; + if result < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(()) + } +} + +fn socklen(length: usize) -> io::Result { + libc::socklen_t::try_from(length).map_err(io::Error::other) +} + +const fn filter_stmt(code: u32, k: u32) -> libc::sock_filter { + filter_jump(code, k, 0, 0) +} + +#[allow( + clippy::cast_possible_truncation, + reason = "classic BPF opcodes are 16-bit by definition" +)] +const fn filter_jump(code: u32, k: u32, jt: u8, jf: u8) -> libc::sock_filter { + libc::sock_filter { + code: code as u16, + jt, + jf, + k, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::{Read as _, Write as _}; + use std::net::{Ipv4Addr, SocketAddr, TcpListener, TcpStream}; + use std::time::Duration; + + #[test] + fn active_probe_passes_without_capabilities() { + probe_loopback_confinement().expect("loopback confinement probe"); + } + + #[test] + fn unbound_socket_reports_no_device() { + let socket = new_socket(libc::AF_INET, libc::SOCK_STREAM).unwrap(); + assert_eq!(bound_device(socket.as_raw_fd()).unwrap(), None); + } + + #[test] + fn confined_socket_cannot_be_rebound() { + let socket = new_socket(libc::AF_INET, libc::SOCK_DGRAM).unwrap(); + confine_to_loopback(socket.as_raw_fd()).unwrap(); + assert!(confine_to_loopback(socket.as_raw_fd()).is_err()); + assert_eq!( + bound_device(socket.as_raw_fd()).unwrap().as_deref(), + Some(&b"lo"[..]) + ); + } + + fn connect_with_timeout(address: SocketAddr) -> io::Result { + TcpStream::connect_timeout(&address, Duration::from_millis(300)) + } + + #[test] + fn loopback_ingress_filter_rejects_loopback_connections() { + let listener = TcpListener::bind((Ipv4Addr::UNSPECIFIED, 0)).unwrap(); + reject_loopback_ingress(listener.as_raw_fd()).unwrap(); + listener.set_nonblocking(true).unwrap(); + let port = listener.local_addr().unwrap().port(); + // Dropped SYNs never complete the handshake. + assert!(connect_with_timeout(SocketAddr::from((Ipv4Addr::LOCALHOST, port))).is_err()); + assert_eq!( + listener.accept().unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + } + + #[test] + fn ingress_filter_admits_other_interfaces() { + // Positive control: the same program keyed to an absent interface + // index must leave loopback traffic untouched. + let listener = TcpListener::bind((Ipv4Addr::LOCALHOST, 0)).unwrap(); + reject_ingress_interface(listener.as_raw_fd(), u32::MAX).unwrap(); + let mut client = connect_with_timeout(listener.local_addr().unwrap()).unwrap(); + let (mut accepted, _) = listener.accept().unwrap(); + client.write_all(b"ping").unwrap(); + let mut buffer = [0_u8; 4]; + accepted.read_exact(&mut buffer).unwrap(); + assert_eq!(&buffer, b"ping"); + } + + #[test] + fn ingress_filter_is_locked() { + let listener = TcpListener::bind((Ipv4Addr::LOCALHOST, 0)).unwrap(); + reject_loopback_ingress(listener.as_raw_fd()).unwrap(); + // SAFETY: SO_DETACH_FILTER ignores its value argument. + let detach = unsafe { + let value = 0_i32; + libc::setsockopt( + listener.as_raw_fd(), + libc::SOL_SOCKET, + libc::SO_DETACH_FILTER, + (&raw const value).cast(), + socklen(size_of::()).unwrap(), + ) + }; + assert!(detach < 0); + } + + /// Opt-in checks against a real non-loopback topology. + /// + /// A trusted harness creates a network namespace whose non-loopback + /// device routes to a peer namespace, enables forwarding and the other + /// permissive routing sysctls, and runs these tests unprivileged inside + /// it. The harness, not the workload, holds any privileges. Each check + /// carries an unconfined positive control so a broken observer cannot + /// pass vacuously. For egress, the harness captures UDP/TCP port 9 on + /// the peer side: exactly one datagram, the unconfined control payload, + /// must arrive. + mod topology { + use super::*; + use std::net::{IpAddr, UdpSocket}; + use std::time::Instant; + + fn env(name: &str) -> String { + std::env::var(name).unwrap_or_else(|_| panic!("{name} is required")) + } + + fn tx_packets(device: &str) -> u64 { + let table = std::fs::read_to_string("/proc/net/dev").unwrap(); + let line = table + .lines() + .find(|line| line.trim_start().starts_with(&format!("{device}:"))) + .unwrap_or_else(|| panic!("{device} is not in this network namespace")); + // Receive has eight columns; transmit packets is the tenth field. + line.split(':') + .nth(1) + .unwrap() + .split_whitespace() + .nth(9) + .unwrap() + .parse() + .unwrap() + } + + fn quiet_counter(device: &str) -> u64 { + let deadline = Instant::now() + Duration::from_secs(5); + loop { + let before = tx_packets(device); + std::thread::sleep(Duration::from_millis(300)); + let after = tx_packets(device); + if before == after { + return after; + } + assert!(Instant::now() < deadline, "{device} never became quiet"); + } + } + + fn confined(domain: i32, kind: i32) -> OwnedFd { + let socket = new_socket(domain, kind | libc::SOCK_NONBLOCK).unwrap(); + confine_to_loopback(socket.as_raw_fd()).unwrap(); + socket + } + + fn attempt_egress(fd: RawFd, kind: i32, destination: SocketAddr) -> Option { + let native = socket2::SockAddr::from(destination); + // SAFETY: payload and the native address are live for each call. + let result = unsafe { + match kind { + libc::SOCK_DGRAM => libc::sendto( + fd, + b"probe".as_ptr().cast(), + 5, + 0, + native.as_ptr().cast(), + native.len(), + ), + _ => libc::sendto( + fd, + b"probe".as_ptr().cast(), + 5, + libc::MSG_FASTOPEN, + native.as_ptr().cast(), + native.len(), + ), + } + }; + (result < 0).then(|| io::Error::last_os_error().raw_os_error().unwrap_or(0)) + } + + #[test] + #[ignore = "requires the privileged topology harness"] + fn topology_confined_sockets_emit_nothing_on_routed_devices() { + let device = env("OPENSHELL_TOPOLOGY_DEVICE"); + let destinations: Vec = env("OPENSHELL_TOPOLOGY_DESTINATIONS") + .split(',') + .map(|value| value.parse().unwrap()) + .collect(); + + // Positive control: an unconfined datagram to the first + // destination is observed on the routed device. + let baseline = quiet_counter(&device); + let unconfined = UdpSocket::bind(match destinations[0].ip() { + IpAddr::V4(_) => "0.0.0.0:0", + IpAddr::V6(_) => "[::]:0", + }) + .unwrap(); + unconfined.send_to(b"control", destinations[0]).unwrap(); + let deadline = Instant::now() + Duration::from_secs(2); + while tx_packets(&device) == baseline { + assert!( + Instant::now() < deadline, + "observer missed the unconfined control" + ); + std::thread::sleep(Duration::from_millis(20)); + } + + let baseline = quiet_counter(&device); + let mut outcomes = Vec::new(); + for destination in &destinations { + let domain = match destination { + SocketAddr::V4(_) => libc::AF_INET, + SocketAddr::V6(_) => libc::AF_INET6, + }; + for kind in [libc::SOCK_DGRAM, libc::SOCK_STREAM] { + let socket = confined(domain, kind); + outcomes.push(( + destination, + kind, + "send", + attempt_egress(socket.as_raw_fd(), kind, *destination), + )); + if kind == libc::SOCK_STREAM { + let socket = confined(domain, kind); + let native = socket2::SockAddr::from(*destination); + // SAFETY: native address is live for the call. + let result = unsafe { + libc::connect(socket.as_raw_fd(), native.as_ptr().cast(), native.len()) + }; + let errno = (result < 0) + .then(|| io::Error::last_os_error().raw_os_error().unwrap_or(0)); + outcomes.push((destination, kind, "connect", errno)); + std::thread::sleep(Duration::from_millis(200)); + } + } + } + std::thread::sleep(Duration::from_millis(500)); + let after = tx_packets(&device); + for outcome in &outcomes { + eprintln!("confined attempt {outcome:?}"); + } + // Link-level chatter (MLD, neighbor discovery) also moves this + // counter, so it cannot prove a negative on its own. The harness + // captures the probe port on the peer side as the authoritative + // observer; report the delta for correlation. + eprintln!("confined phase {device} tx delta {}", after - baseline); + } + + #[test] + #[ignore = "requires the privileged topology harness"] + fn topology_confined_listeners_reject_routed_ingress() { + let confined_port: u16 = env("OPENSHELL_TOPOLOGY_CONFINED_PORT").parse().unwrap(); + let control_port: u16 = env("OPENSHELL_TOPOLOGY_CONTROL_PORT").parse().unwrap(); + let wait = Duration::from_secs(env("OPENSHELL_TOPOLOGY_WAIT_SECS").parse().unwrap()); + let listen = |port: u16, confine: bool| { + let socket = + socket2::Socket::new(socket2::Domain::IPV6, socket2::Type::STREAM, None) + .unwrap(); + // Dual-stack wildcard covers IPv4 and IPv4-mapped peers too. + socket.set_only_v6(false).unwrap(); + if confine { + confine_to_loopback(socket.as_raw_fd()).unwrap(); + } + socket + .bind(&SocketAddr::from((std::net::Ipv6Addr::UNSPECIFIED, port)).into()) + .unwrap(); + socket.listen(16).unwrap(); + socket.set_nonblocking(true).unwrap(); + TcpListener::from(socket) + }; + let confined_listener = listen(confined_port, true); + let control_listener = listen(control_port, false); + let deadline = Instant::now() + wait; + let (mut confined_peers, mut control_peers) = (Vec::new(), Vec::new()); + while Instant::now() < deadline { + while let Ok((_, peer)) = confined_listener.accept() { + confined_peers.push(peer); + } + while let Ok((_, peer)) = control_listener.accept() { + control_peers.push(peer); + } + std::thread::sleep(Duration::from_millis(20)); + } + eprintln!("control accepted {control_peers:?}; confined accepted {confined_peers:?}"); + assert!( + control_peers.iter().any(|peer| !peer.ip().is_loopback() + && peer.ip() != IpAddr::from(std::net::Ipv6Addr::LOCALHOST)), + "positive control saw no routed client" + ); + assert!( + confined_peers.iter().all(|peer| match peer.ip() { + IpAddr::V4(ip) => ip.is_loopback(), + IpAddr::V6(ip) => + ip.is_loopback() || ip.to_ipv4_mapped().is_some_and(|ip| ip.is_loopback()), + }), + "confined listener accepted a routed client" + ); + } + } +} diff --git a/crates/openshell-isolation-interface/src/linux/socket_registry.rs b/crates/openshell-isolation-interface/src/linux/socket_registry.rs index 7a245d1eac..9754720792 100644 --- a/crates/openshell-isolation-interface/src/linux/socket_registry.rs +++ b/crates/openshell-isolation-interface/src/linux/socket_registry.rs @@ -76,8 +76,6 @@ pub enum SocketState { DnsTcp { relay: SocketAddr }, /// Workload-owned listening socket. Listening { local: SocketAddr }, - /// Stream accepted from a verified local peer. - AcceptedLocal { peer: SocketAddr }, /// A committed relay failed after connection. Failed { errno: i32 }, } @@ -254,9 +252,8 @@ impl SocketRegistry { /// Publish a tentative socket in a caller-proven initial state. /// - /// Accepted sockets are created and classified by the trusted broker, so - /// they enter the registry directly as [`SocketState::AcceptedLocal`] - /// rather than pretending to be unconnected. + /// Used when the trusted broker has already established the socket's + /// state before publication, so the entry never appears unconnected. pub fn commit_with_state( &mut self, tentative: TentativeSocket, diff --git a/crates/openshell-sandbox-backend/src/boundary_protocol.rs b/crates/openshell-sandbox-backend/src/boundary_protocol.rs index 6c6c66e76c..5b46adbff1 100644 --- a/crates/openshell-sandbox-backend/src/boundary_protocol.rs +++ b/crates/openshell-sandbox-backend/src/boundary_protocol.rs @@ -124,6 +124,10 @@ pub struct NativeLinuxSandboxAuditEvidence { pub tcp_dns_round_trip: bool, pub tcp_allow_round_trip: bool, pub tcp_deny_round_trip: bool, + /// Workload INET sockets are bound to loopback before injection, the + /// binding cannot be changed from sandbox credentials, and accepted + /// sockets inherit it. Native local `accept` depends on this property. + pub socket_loopback_confinement: bool, } impl NativeLinuxSandboxAuditEvidence { @@ -150,7 +154,8 @@ impl NativeLinuxSandboxAuditEvidence { && self.udp_dns_round_trip && self.tcp_dns_round_trip && self.tcp_allow_round_trip - && self.tcp_deny_round_trip; + && self.tcp_deny_round_trip + && self.socket_loopback_confinement; if complete { Ok(()) } else { @@ -175,7 +180,8 @@ impl NativeLinuxSandboxAuditEvidence { && self.udp_dns_round_trip && self.tcp_dns_round_trip && self.tcp_allow_round_trip - && self.tcp_deny_round_trip, + && self.tcp_deny_round_trip + && self.socket_loopback_confinement, "seccomp-notify", ), request_attribution: EnforcedProperty::new( @@ -1465,6 +1471,7 @@ mod tests { tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, } } @@ -1488,6 +1495,24 @@ mod tests { assert!(!audit.properties().egress_interception.enforced); } + #[test] + fn audit_evidence_requires_socket_loopback_confinement() { + let mut audit = complete_audit_evidence(); + audit.socket_loopback_confinement = false; + assert!(audit.validate().is_err()); + assert!(!audit.properties().egress_interception.enforced); + } + + #[test] + fn audit_evidence_rejects_missing_socket_loopback_confinement_field() { + let mut value = serde_json::to_value(complete_audit_evidence()).unwrap(); + value + .as_object_mut() + .unwrap() + .remove("socket_loopback_confinement"); + assert!(serde_json::from_value::(value).is_err()); + } + #[test] fn audit_evidence_accepts_legacy_read_only_listener() { let mut audit = complete_audit_evidence(); diff --git a/crates/openshell-sandbox-backend/src/runtime.rs b/crates/openshell-sandbox-backend/src/runtime.rs index 641d50cd24..c1e36f1b4c 100644 --- a/crates/openshell-sandbox-backend/src/runtime.rs +++ b/crates/openshell-sandbox-backend/src/runtime.rs @@ -2916,6 +2916,7 @@ mod tests { tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, }; openshell_isolation_interface::contract::BoundaryConfirmation { generation: "test-generation".to_string(), diff --git a/crates/openshell-sandbox/src/accept_interrupt.rs b/crates/openshell-sandbox/src/accept_interrupt.rs deleted file mode 100644 index 86bbf39218..0000000000 --- a/crates/openshell-sandbox/src/accept_interrupt.rs +++ /dev/null @@ -1,297 +0,0 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -//! Cancellation for broker-owned blocking accepts without changing workload OFDs. -//! -//! SIGUSR2 is reserved by the sandbox binary. Its process-global disposition is -//! necessarily kernel state, not a global application context. All registration, -//! cancellation and thread ownership state belongs to one broker instance. - -#![allow(unsafe_code)] - -use std::collections::HashMap; -use std::io; -use std::marker::PhantomData; -use std::rc::Rc; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Condvar, Mutex}; -use std::time::Duration; - -const INTERRUPT_SIGNAL: libc::c_int = libc::SIGUSR2; -const INTERRUPT_INTERVAL: Duration = Duration::from_millis(10); - -extern "C" fn interrupt_accept(_: libc::c_int) {} - -fn reserve_signal() -> io::Result<()> { - // SAFETY: both actions are initialized storage. The no-op handler is - // async-signal-safe and deliberately omits SA_RESTART so accept returns EINTR. - unsafe { - let mut previous: libc::sigaction = std::mem::zeroed(); - if libc::sigaction(INTERRUPT_SIGNAL, std::ptr::null(), &raw mut previous) < 0 { - return Err(io::Error::last_os_error()); - } - if previous.sa_sigaction != libc::SIG_DFL - && previous.sa_sigaction != interrupt_accept as *const () as usize - { - return Err(io::Error::other( - "sandbox SIGUSR2 is already reserved by another handler", - )); - } - let mut action: libc::sigaction = std::mem::zeroed(); - action.sa_sigaction = interrupt_accept as *const () as usize; - libc::sigemptyset(&raw mut action.sa_mask); - if libc::sigaction(INTERRUPT_SIGNAL, &raw const action, std::ptr::null_mut()) < 0 { - return Err(io::Error::last_os_error()); - } - } - Ok(()) -} - -#[derive(Default)] -struct State { - workers: Mutex>, - changed: Condvar, - stopped: AtomicBool, -} - -// musl represents pthread_t as an opaque pointer, unlike glibc's integer. It -// is only passed back to pthread_kill, never dereferenced by this module. -struct RegisteredThread(libc::pthread_t); - -// SAFETY: POSIX permits signaling a live pthread from another thread. The -// handle is accessed only under State::workers, and the owning worker removes -// its registration under that same mutex before returning. AcceptRegistration -// cannot move to another thread, so its Drop cannot outlive the owning worker. -unsafe impl Send for RegisteredThread {} - -pub struct AcceptMonitor { - state: Arc, - thread: Option>, -} - -impl AcceptMonitor { - pub(crate) fn start(valid: impl Fn(u64) -> bool + Send + 'static) -> io::Result { - reserve_signal()?; - let state = Arc::new(State::default()); - let worker_state = state.clone(); - let thread = std::thread::Builder::new() - .name("openshell-accept-cancellation".into()) - .spawn(move || monitor(&worker_state, valid))?; - Ok(Self { - state, - thread: Some(thread), - }) - } - - pub(crate) fn registrar(&self) -> AcceptRegistrar { - AcceptRegistrar(self.state.clone()) - } -} - -impl Drop for AcceptMonitor { - fn drop(&mut self) { - let workers = lock(&self.state.workers); - self.state.stopped.store(true, Ordering::Release); - self.state.changed.notify_all(); - drop(workers); - // The monitor keeps interrupting registered workers during shutdown. - // Registrations are removed before their threads can exit/reuse IDs. - if let Some(thread) = self.thread.take() { - let _ = thread.join(); - } - } -} - -#[derive(Clone)] -pub struct AcceptRegistrar(Arc); - -impl AcceptRegistrar { - pub(crate) fn register(&self, notification_id: u64) -> io::Result { - // SAFETY: this changes only the current broker worker's signal mask. - // Workload launchers do not inherit this mask; exec resets the handler. - let thread = unsafe { - let mut mask: libc::sigset_t = std::mem::zeroed(); - libc::sigemptyset(&raw mut mask); - libc::sigaddset(&raw mut mask, INTERRUPT_SIGNAL); - let error = - libc::pthread_sigmask(libc::SIG_UNBLOCK, &raw const mask, std::ptr::null_mut()); - if error != 0 { - return Err(io::Error::from_raw_os_error(error)); - } - libc::pthread_self() - }; - let mut workers = lock(&self.0.workers); - if self.0.stopped.load(Ordering::Acquire) { - return Err(io::Error::from_raw_os_error(libc::ECANCELED)); - } - if workers.contains_key(¬ification_id) { - return Err(io::Error::other( - "duplicate accept notification registration", - )); - } - workers.insert(notification_id, RegisteredThread(thread)); - self.0.changed.notify_one(); - Ok(AcceptRegistration { - state: self.0.clone(), - notification_id, - owning_thread: PhantomData, - }) - } -} - -pub struct AcceptRegistration { - state: Arc, - notification_id: u64, - // Drop must run on the registering thread before its pthread_t can expire. - // No Rc is allocated; this marker makes the guard neither Send nor Sync. - owning_thread: PhantomData>, -} - -impl AcceptRegistration { - pub(crate) fn ensure_running(&self) -> io::Result<()> { - if self.state.stopped.load(Ordering::Acquire) { - Err(io::Error::from_raw_os_error(libc::ECANCELED)) - } else { - Ok(()) - } - } -} - -impl Drop for AcceptRegistration { - fn drop(&mut self) { - lock(&self.state.workers).remove(&self.notification_id); - self.state.changed.notify_one(); - } -} - -fn monitor(state: &State, valid: impl Fn(u64) -> bool) { - let mut workers = lock(&state.workers); - loop { - let stopped = state.stopped.load(Ordering::Acquire); - if stopped && workers.is_empty() { - return; - } - for (¬ification_id, thread) in &*workers { - if stopped || !valid(notification_id) { - // SAFETY: the registration lock pins this live pthread_t. - // Repeated interrupts close the check-to-accept race: a signal - // received before accept cannot leave a later accept stranded. - let _ = unsafe { libc::pthread_kill(thread.0, INTERRUPT_SIGNAL) }; - } - } - workers = if workers.is_empty() { - state - .changed - .wait(workers) - .unwrap_or_else(std::sync::PoisonError::into_inner) - } else { - state - .changed - .wait_timeout(workers, INTERRUPT_INTERVAL) - .unwrap_or_else(std::sync::PoisonError::into_inner) - .0 - }; - } -} - -fn lock(mutex: &Mutex) -> std::sync::MutexGuard<'_, T> { - mutex - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) -} - -#[cfg(test)] -mod tests { - use super::*; - use std::net::{TcpListener, TcpStream}; - use std::os::fd::AsRawFd; - - #[test] - fn registrar_crosses_threads_but_registration_ends_before_worker_exit() { - fn assert_send_sync() {} - assert_send_sync::(); - - let monitor = AcceptMonitor::start(|_| true).unwrap(); - let registrar = monitor.registrar(); - std::thread::spawn(move || { - let registration = registrar.register(3).unwrap(); - assert!(registrar.register(3).is_err()); - assert!(lock(®istrar.0.workers).contains_key(&3)); - drop(registration); - assert!(lock(®istrar.0.workers).is_empty()); - }) - .join() - .unwrap(); - assert!(lock(&monitor.state.workers).is_empty()); - } - - #[test] - fn cancellation_interrupts_competing_accept_after_readiness_was_consumed() { - let valid = Arc::new(AtomicBool::new(true)); - let monitored = valid.clone(); - let monitor = AcceptMonitor::start(move |_| monitored.load(Ordering::Acquire)).unwrap(); - let listener = TcpListener::bind("127.0.0.1:0").unwrap(); - let client = TcpStream::connect(listener.local_addr().unwrap()).unwrap(); - // Both contenders could observe this same readable listener. Consume - // its only connection before the second contender actually accepts. - let accepted = listener.accept().unwrap(); - let registrar = monitor.registrar(); - let (ready_tx, ready_rx) = std::sync::mpsc::channel(); - let (done_tx, done_rx) = std::sync::mpsc::channel(); - let worker = std::thread::spawn(move || { - let registration = registrar.register(1).unwrap(); - ready_tx.send(()).unwrap(); - // SAFETY: the listener is live and null address outputs are valid. - // Use the syscall directly: std::net retries EINTR internally. - let result = unsafe { - libc::accept4( - listener.as_raw_fd(), - std::ptr::null_mut(), - std::ptr::null_mut(), - libc::SOCK_CLOEXEC, - ) - }; - assert_eq!(result, -1); - let error = io::Error::last_os_error(); - assert_eq!(error.kind(), io::ErrorKind::Interrupted); - drop(registration); - done_tx.send(()).unwrap(); - }); - ready_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - valid.store(false, Ordering::Release); - done_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - worker.join().unwrap(); - drop((accepted, client, monitor)); - } - - #[test] - fn shutdown_interrupts_registered_accepts_and_reclaims_the_monitor() { - let monitor = AcceptMonitor::start(|_| true).unwrap(); - let registrar = monitor.registrar(); - let listener = TcpListener::bind("127.0.0.1:0").unwrap(); - let (ready_tx, ready_rx) = std::sync::mpsc::channel(); - let (done_tx, done_rx) = std::sync::mpsc::channel(); - let worker = std::thread::spawn(move || { - let registration = registrar.register(2).unwrap(); - ready_tx.send(()).unwrap(); - // SAFETY: owned listener and optional null address outputs. - let result = unsafe { - libc::accept4( - listener.as_raw_fd(), - std::ptr::null_mut(), - std::ptr::null_mut(), - libc::SOCK_CLOEXEC, - ) - }; - assert_eq!(result, -1); - assert!(registration.ensure_running().is_err()); - drop(registration); - done_tx.send(()).unwrap(); - }); - ready_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - let shutdown = std::thread::spawn(move || drop(monitor)); - done_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - worker.join().unwrap(); - shutdown.join().unwrap(); - } -} diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index f84f98d424..e271fd4961 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -2406,6 +2406,7 @@ mod linux { tcp_dns_round_trip: self.qualification.tcp_dns_round_trip, tcp_allow_round_trip: self.qualification.tcp_allow_round_trip, tcp_deny_round_trip: self.qualification.tcp_deny_round_trip, + socket_loopback_confinement: self.qualification.socket_loopback_confinement, }; // The boundary reports mechanism evidence; the authenticated host // backend validates it before constructing a ConfirmedBoundary. @@ -3216,7 +3217,7 @@ mod linux { }) } BoundaryListenerConfig::TlsTcp { address, tls } => { - let listener = std::net::TcpListener::bind(address)?; + let listener = Self::bind_tcp(*address)?; listener.set_nonblocking(true)?; let server_config = Arc::new(load_tls_server_config(tls)?); Ok(Self::Tcp { @@ -3227,6 +3228,33 @@ mod linux { } } + /// Bind the TCP control listener. + /// + /// A listener on a non-loopback address serves a supervisor in another + /// network namespace, so it rejects all loopback-interface ingress + /// before it starts listening. Workload sockets are bound to loopback + /// and natively accepted ones are not registered with the broker, so + /// this standing filter, not the broker's port reservation, keeps the + /// workload from reaching the control endpoint through loopback or + /// the pod's own address. + fn bind_tcp(address: std::net::SocketAddr) -> io::Result { + let socket = socket2::Socket::new( + socket2::Domain::for_address(address), + socket2::Type::STREAM, + Some(socket2::Protocol::TCP), + )?; + socket.set_cloexec(true)?; + socket.set_reuse_address(true)?; + if !address.ip().is_loopback() { + openshell_isolation_interface::linux::socket_confinement::reject_loopback_ingress( + socket.as_raw_fd(), + )?; + } + socket.bind(&address.into())?; + socket.listen(128)?; + Ok(socket.into()) + } + fn bind_vsock(port: u32) -> io::Result { let family = libc::sa_family_t::try_from(libc::AF_VSOCK).map_err(|_| { io::Error::new(io::ErrorKind::InvalidInput, "AF_VSOCK exceeds sa_family_t") @@ -4511,6 +4539,7 @@ mod linux { tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, } } @@ -4767,6 +4796,35 @@ mod linux { server.abort(); } + #[test] + fn pod_control_listener_rejects_loopback_ingress() { + let directory = tempfile::tempdir().expect("temporary directory"); + let (server_tls, _client_tls) = stage_test_tls(directory.path(), "loopback"); + let listener = ControlListener::bind(&BoundaryListenerConfig::TlsTcp { + address: "0.0.0.0:0".parse().expect("valid address"), + tls: server_tls, + }) + .expect("bind TLS listener"); + let port = listener + .tcp_local_addr() + .expect("TLS listener address") + .port(); + // Loopback and the host's own address both arrive on `lo`; the + // dropped SYN never completes a handshake. + let result = std::net::TcpStream::connect_timeout( + &std::net::SocketAddr::from(([127, 0, 0, 1], port)), + Duration::from_millis(300), + ); + assert!( + result.is_err(), + "loopback client reached the control listener" + ); + assert!(matches!( + listener.accept().map(|_| ()), + Err(error) if error.kind() == io::ErrorKind::WouldBlock + )); + } + #[test] fn tls_listener_preserves_session_when_control_switches_to_async_streaming() { let directory = tempfile::tempdir().expect("temporary directory"); diff --git a/crates/openshell-sandbox/src/lib.rs b/crates/openshell-sandbox/src/lib.rs index a8d31fbfe9..110a395842 100644 --- a/crates/openshell-sandbox/src/lib.rs +++ b/crates/openshell-sandbox/src/lib.rs @@ -3,8 +3,6 @@ //! Capability-free in-workload sandbox boundary. -#[cfg(target_os = "linux")] -mod accept_interrupt; pub mod boundary_exec; pub mod boundary_io; mod boundary_server; @@ -44,6 +42,7 @@ pub struct RuntimeQualification { pub tcp_dns_round_trip: bool, pub tcp_allow_round_trip: bool, pub tcp_deny_round_trip: bool, + pub socket_loopback_confinement: bool, } /// Placeholder used when compiling the package on a non-Linux host. diff --git a/crates/openshell-sandbox/src/main.rs b/crates/openshell-sandbox/src/main.rs index 99f37b202e..aaec0a0d11 100644 --- a/crates/openshell-sandbox/src/main.rs +++ b/crates/openshell-sandbox/src/main.rs @@ -95,6 +95,9 @@ struct QualificationReport { task_memory_copy: bool, connected_send_fast_path: bool, socket_virtualization: bool, + /// Workload INET sockets are bound to loopback, the binding cannot be + /// changed from sandbox credentials, and accepted sockets inherit it. + socket_loopback_confinement: bool, dns_relay_bind: bool, udp_dns_round_trip: bool, tcp_dns_round_trip: bool, @@ -163,6 +166,9 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi .into_diagnostic() .wrap_err("seccomp notification probe")?; probe_socket_virtualization().wrap_err("socket virtualization probe")?; + openshell_isolation_interface::linux::socket_confinement::probe_loopback_confinement() + .into_diagnostic() + .wrap_err("socket loopback confinement probe")?; probe_dns_relay_bind().wrap_err("DNS relay bind probe")?; let landlock_abi = openshell_isolation_interface::linux::landlock::abi_version() .into_diagnostic() @@ -196,6 +202,7 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi task_memory_copy, connected_send_fast_path: notification.connected_send_fast_path(), socket_virtualization: true, + socket_loopback_confinement: true, dns_relay_bind: true, udp_dns_round_trip: true, tcp_dns_round_trip: true, @@ -230,6 +237,7 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, }; Ok((qualification, report)) } diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index f2f196c925..c60e061a2a 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -30,11 +30,9 @@ use tokio::sync::{mpsc, oneshot}; const SOCKET_CAPACITY: usize = 4_096; const SOCKET_FD_HEADROOM: usize = 64; const OPEN_QUEUE_CAPACITY: usize = 256; -const ACCEPT_WORKER_CAPACITY: usize = 64; const DNS_QUEUE_CAPACITY: usize = 256; const DNS_WORKER_CAPACITY: usize = 256; const DNS_QUERY_TIMEOUT: Duration = Duration::from_secs(10); -const ACCEPT_POLL_INTERVAL: Duration = Duration::from_millis(250); const DNS_RELAY_ADDRESS: SocketAddr = SocketAddr::V4(std::net::SocketAddrV4::new( Ipv4Addr::new(127, 0, 0, 53), 53, @@ -73,27 +71,6 @@ fn acquire_pending_dns_slot(active: &Arc) -> io::Result, -} - -impl Drop for PendingAcceptSlot { - fn drop(&mut self) { - self.active.fetch_sub(1, Ordering::AcqRel); - } -} - -fn acquire_pending_accept_slot(active: &Arc) -> io::Result { - active - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { - (current < ACCEPT_WORKER_CAPACITY).then_some(current + 1) - }) - .map_err(|_| io::Error::from_raw_os_error(libc::EAGAIN))?; - Ok(PendingAcceptSlot { - active: Arc::clone(active), - }) -} - fn acquire_pending_open_slot(active: &Arc) -> io::Result { active .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { @@ -192,12 +169,10 @@ fn register_dns_socket( struct NotificationQueues { provider_files: crate::provider_files::ProviderFiles, protected_control_port: Option, - accept_registrar: crate::accept_interrupt::AcceptRegistrar, identity_resolver: ProcfsIdentityResolver, pending: mpsc::Sender, dns_relay: DnsRelay, active_opens: Arc, - active_accepts: Arc, retained_socket_capacity: usize, decision_timeout: Duration, } @@ -206,7 +181,6 @@ struct NotificationQueues { #[derive(Clone)] pub struct NetworkBroker { provider_files: crate::provider_files::ProviderFiles, - _accept_monitor: Arc, pending: Arc>>, pending_dns: Arc>>, dns_address: SocketAddr, @@ -250,14 +224,9 @@ impl NetworkBroker { decision_timeout: Duration, ) -> io::Result { let listener = Arc::new(listener); - let monitor_listener = listener.clone(); - let accept_monitor = Arc::new(crate::accept_interrupt::AcceptMonitor::start(move |id| { - monitor_listener.validate_id(id).is_ok() - })?); let (pending_tx, pending_rx) = mpsc::channel(OPEN_QUEUE_CAPACITY); let (pending_dns_tx, pending_dns_rx) = mpsc::channel(DNS_QUEUE_CAPACITY); let active_opens = Arc::new(AtomicUsize::new(0)); - let active_accepts = Arc::new(AtomicUsize::new(0)); let dns_relay = start_dns_relay(dns_address, pending_dns_tx)?; let dns_address = dns_relay.address; let retained_socket_capacity = retained_socket_capacity()?; @@ -266,12 +235,10 @@ impl NetworkBroker { let queues = NotificationQueues { provider_files: provider_files.clone(), protected_control_port, - accept_registrar: accept_monitor.registrar(), identity_resolver: ProcfsIdentityResolver::for_pid_namespace(), pending: pending_tx, dns_relay, active_opens, - active_accepts, retained_socket_capacity, decision_timeout, }; @@ -316,7 +283,6 @@ impl NetworkBroker { .map_err(|error| io::Error::other(format!("start network broker: {error}")))?; Ok(Self { provider_files, - _accept_monitor: accept_monitor, pending: Arc::new(tokio::sync::Mutex::new(pending_rx)), pending_dns: Arc::new(tokio::sync::Mutex::new(pending_dns_rx)), dns_address, @@ -575,15 +541,6 @@ fn dispatch_notification( if syscall == libc::SYS_listen { return listen_socket(®istry, &listener, notification); } - if matches!(syscall, libc::SYS_accept | libc::SYS_accept4) { - return accept_socket( - registry, - listener, - notification, - queues.active_accepts, - queues.accept_registrar, - ); - } if matches!( syscall, libc::SYS_sendto | libc::SYS_sendmsg | libc::SYS_sendmmsg @@ -598,9 +555,7 @@ fn dispatch_notification( .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; let option = i32::try_from(notification.args[2]) .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; - if (level == libc::IPPROTO_TCP && option == libc::TCP_FASTOPEN_CONNECT) - || (level == libc::IPPROTO_IPV6 && option == libc::IPV6_ADDRFORM) - { + if socket_option_is_denied(level, option) { return Err(io::Error::from_raw_os_error(libc::EPERM)); } return listener.respond_continue(notification.id); @@ -608,6 +563,32 @@ fn dispatch_notification( Err(io::Error::from_raw_os_error(libc::EPERM)) } +/// Options the workload may never set, decided from scalar syscall arguments +/// that another thread cannot replace before the kernel reads them. +/// +/// Interface-selection options could redirect or unpin a socket's loopback +/// device binding. The kernel already refuses to change an existing binding +/// without `CAP_NET_RAW`; denying them here keeps confinement independent of +/// the capability state of the namespace that owns the network namespace. +fn socket_option_is_denied(level: i32, option: i32) -> bool { + matches!( + (level, option), + (libc::IPPROTO_TCP, libc::TCP_FASTOPEN_CONNECT) + | ( + libc::IPPROTO_IPV6, + libc::IPV6_ADDRFORM | libc::IPV6_UNICAST_IF | libc::IPV6_MULTICAST_IF + ) + | ( + libc::SOL_SOCKET, + libc::SO_BINDTODEVICE | libc::SO_BINDTOIFINDEX + ) + | ( + libc::IPPROTO_IP, + libc::IP_UNICAST_IF | libc::IP_MULTICAST_IF + ) + ) +} + fn create_socket( registry: &Mutex, listener: &NotificationListener, @@ -654,6 +635,12 @@ fn create_socket( } // SAFETY: successful socket returned one owned descriptor. let source = unsafe { OwnedFd::from_raw_fd(source) }; + // Confinement is standing kernel state that must exist before the workload + // can observe the descriptor. Natively accepted children inherit it, so + // local accept needs no per-connection broker inspection. + openshell_isolation_interface::linux::socket_confinement::confine_to_loopback( + source.as_raw_fd(), + )?; let metadata = SocketMetadata { family, kind, @@ -1091,218 +1078,6 @@ fn listen_socket( listener.respond_value(notification.id, 0) } -fn accept_socket( - registry: Arc>, - listener: Arc, - notification: Notification, - active_accepts: Arc, - accept_registrar: crate::accept_interrupt::AcceptRegistrar, -) -> io::Result<()> { - let fd = raw_fd(notification.args[0])?; - let flags = if i64::from(notification.syscall) == libc::SYS_accept4 { - i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))? - } else { - 0 - }; - if flags & !(libc::SOCK_CLOEXEC | libc::SOCK_NONBLOCK) != 0 { - return Err(io::Error::from_raw_os_error(libc::EINVAL)); - } - if (notification.args[1] == 0) != (notification.args[2] == 0) { - return Err(io::Error::from_raw_os_error(libc::EFAULT)); - } - let (listener_inode, metadata, source) = { - let registry = lock(®istry); - let Ok(entry) = registry.resolve(notification.tid, fd) else { - return listener.respond_continue(notification.id); - }; - if !matches!(entry.state(), SocketState::Listening { .. }) - || entry.metadata().kind != InetKind::Tcp - { - return Err(io::Error::from_raw_os_error(libc::EINVAL)); - } - let source = duplicate_close_on_exec(entry.retained_preconnect()?.as_raw_fd())?; - (entry.identity().inode, entry.metadata(), source) - }; - let slot = acquire_pending_accept_slot(&active_accepts)?; - let worker_listener = Arc::clone(&listener); - std::thread::Builder::new() - .name("openshell-local-accept".to_string()) - .spawn(move || { - let _slot = slot; - let registration = match accept_registrar.register(notification.id) { - Ok(registration) => registration, - Err(error) => { - let _ = worker_listener.respond_errno(notification.id, error_to_errno(&error)); - return; - } - }; - if let Err(error) = accept_and_inject( - ®istry, - &worker_listener, - notification, - AcceptOperation { - flags, - listener_inode, - metadata, - source, - registration, - }, - ) { - let _ = worker_listener.respond_errno(notification.id, error_to_errno(&error)); - } - }) - .map_err(|error| io::Error::other(format!("start local-accept worker: {error}")))?; - Ok(()) -} - -struct AcceptOperation { - flags: i32, - listener_inode: u64, - metadata: SocketMetadata, - source: OwnedFd, - registration: crate::accept_interrupt::AcceptRegistration, -} - -fn accept_and_inject( - registry: &Mutex, - listener: &NotificationListener, - notification: Notification, - operation: AcceptOperation, -) -> io::Result<()> { - let AcceptOperation { - flags, - listener_inode, - metadata, - source, - registration, - } = operation; - let mut poll = libc::pollfd { - fd: source.as_raw_fd(), - events: libc::POLLIN, - revents: 0, - }; - // SAFETY: F_GETFL reads the live listener OFD flags. - let current_flags = unsafe { libc::fcntl(source.as_raw_fd(), libc::F_GETFL) }; - if current_flags < 0 { - return Err(io::Error::last_os_error()); - } - let nonblocking = current_flags & libc::O_NONBLOCK != 0; - let timeout = if nonblocking { - 0 - } else { - i32::try_from(ACCEPT_POLL_INTERVAL.as_millis()).map_err(io::Error::other)? - }; - // Readiness may disappear before accept (another accept or an aborted - // connection). The registered watchdog interrupts a blocked syscall when - // its notification dies or the broker shuts down. No workload OFD flags - // are changed, and no worker can outlive its cancellation registration. - loop { - registration.ensure_running()?; - listener.validate_id(notification.id)?; - // SAFETY: poll references one live pollfd for this call. - let ready = unsafe { libc::poll(&raw mut poll, 1, timeout) }; - if ready < 0 { - let error = io::Error::last_os_error(); - if error.kind() == io::ErrorKind::Interrupted { - continue; - } - return Err(error); - } - if ready == 0 { - if nonblocking { - return Err(io::Error::from_raw_os_error(libc::EAGAIN)); - } - continue; - } - break; - } - - let mut storage = std::mem::MaybeUninit::::zeroed(); - let mut length = - libc::socklen_t::try_from(size_of::()).map_err(io::Error::other)?; - // Always keep the broker-side descriptor close-on-exec. ADDFD separately - // applies the workload's requested descriptor flag. - let accepted_flags = flags | libc::SOCK_CLOEXEC; - // SAFETY: storage and length are live outputs and source is a listening - // socket proven by the registry. - let accepted = unsafe { - libc::accept4( - source.as_raw_fd(), - storage.as_mut_ptr().cast(), - &raw mut length, - accepted_flags, - ) - }; - if accepted < 0 { - return Err(io::Error::last_os_error()); - } - // Only the blocking accept phase needs asynchronous interruption. Stop - // monitoring before ADDFD completes the notification, otherwise a normal - // successful response could be mistaken for cancellation during commit. - drop(registration); - // SAFETY: successful accept4 returned one newly owned descriptor. - let accepted = unsafe { OwnedFd::from_raw_fd(accepted) }; - // SAFETY: accept4 initialized the reported prefix of storage. - let peer = decode_sockaddr( - unsafe { storage.assume_init() }, - usize::try_from(length).unwrap_or(0), - )?; - if !peer.ip().is_loopback() { - return Err(io::Error::from_raw_os_error(libc::EACCES)); - } - if notification.args[1] != 0 { - write_socket_addr( - listener, - notification.id, - notification.tid, - notification.args[1], - notification.args[2], - peer, - )?; - } - - let accepted_metadata = SocketMetadata { - family: metadata.family, - kind: InetKind::Tcp, - close_on_exec: flags & libc::SOCK_CLOEXEC != 0, - nonblocking: flags & libc::SOCK_NONBLOCK != 0, - creator_generation: u64::from(notification.tid), - }; - let mut registry = lock(registry); - let notifying_fd = raw_fd(notification.args[0])?; - if registry - .resolve(notification.tid, notifying_fd)? - .identity() - .inode - != listener_inode - { - return Err(io::Error::from_raw_os_error(libc::EBADF)); - } - if registry.is_full() { - collect_closed_socket_entries_locked(&mut registry)?; - } - let tentative = registry.stage(accepted, accepted_metadata)?; - listener.add_fd_and_send( - notification.id, - tentative.source_fd(), - accepted_metadata.close_on_exec, - )?; - registry.commit_with_state(tentative, SocketState::AcceptedLocal { peer })?; - Ok(()) -} - -fn duplicate_close_on_exec(fd: RawFd) -> io::Result { - // SAFETY: F_DUPFD_CLOEXEC returns an independent owned descriptor for the - // same open-file description. - let duplicate = unsafe { libc::fcntl(fd, libc::F_DUPFD_CLOEXEC, 3) }; - if duplicate < 0 { - return Err(io::Error::last_os_error()); - } - // SAFETY: successful fcntl returned one newly owned descriptor. - Ok(unsafe { OwnedFd::from_raw_fd(duplicate) }) -} - fn classify_send( registry: &Mutex, listener: &NotificationListener, @@ -1311,6 +1086,12 @@ fn classify_send( ) -> io::Result<()> { let fd = raw_fd(notification.args[0])?; let syscall = i64::from(notification.syscall); + // Fast Open turns a send into a connect. Decide from the scalar flags + // argument, which another thread cannot replace, so the denial also + // covers natively accepted and other unregistered descriptors. + if send_flags(syscall, notification.args) & libc::MSG_FASTOPEN != 0 { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } let (state, metadata) = { let registry = lock(registry); let Ok(entry) = registry.resolve(notification.tid, fd) else { @@ -1320,10 +1101,8 @@ fn classify_send( }; (entry.state().clone(), entry.metadata()) }; - if matches!( - &state, - SocketState::Connected { .. } | SocketState::AcceptedLocal { .. } - ) || (metadata.kind == InetKind::Tcp && matches!(&state, SocketState::Local { .. })) + if matches!(&state, SocketState::Connected { .. }) + || (metadata.kind == InetKind::Tcp && matches!(&state, SocketState::Local { .. })) { return listener.respond_continue(notification.id); } @@ -1419,12 +1198,29 @@ fn classify_send( listener.respond_value(notification.id, result) } Ok(_) => Err(io::Error::from_raw_os_error(libc::EDESTADDRREQ)), - // Non-INET sockets and accepted local sockets were never registered. - // The mandatory outer fence still prevents an external kernel route. + // Non-INET sockets and natively accepted sockets were never + // registered. Accepted sockets inherit their listener's loopback + // binding, and the mandatory outer fence remains an independent + // backstop against an external kernel route. Err(_) => listener.respond_continue(notification.id), } } +fn send_flags(syscall: i64, args: [u64; 6]) -> i32 { + let flags = match syscall { + libc::SYS_sendmsg => args[2], + libc::SYS_sendto | libc::SYS_sendmmsg => args[3], + _ => 0, + }; + // Syscall flag arguments are C ints; the kernel ignores the upper word. + #[allow( + clippy::cast_possible_truncation, + reason = "the kernel reads only the low 32 bits of the flags argument" + )] + let flags = flags as u32; + flags.cast_signed() +} + struct SendMessage { data: Vec, destination: Option, @@ -1583,11 +1379,14 @@ fn get_peer_name( let Ok(entry) = registry.resolve(notification.tid, fd) else { return listener.respond_continue(notification.id); }; - let peer = match entry.state() { - SocketState::Connected { original_peer } => *original_peer, - SocketState::Local { peer } | SocketState::AcceptedLocal { peer } => *peer, - _ => return Err(io::Error::from_raw_os_error(libc::ENOTCONN)), + // Only relayed sockets need a synthesized original peer. Every other + // descriptor reports its true kernel peer, including on legacy listeners + // where the broker cannot write into workload memory. Continuing on a + // substituted descriptor only discloses that descriptor's own peer. + let SocketState::Connected { original_peer } = entry.state() else { + return listener.respond_continue(notification.id); }; + let peer = *original_peer; write_socket_addr( listener, notification.id, @@ -1754,11 +1553,10 @@ fn write_socket_addr( value: SocketAddr, ) -> io::Result<()> { // A LegacyReadOnly listener (kernels < 5.19) cannot safely write into - // workload memory: without WAIT_KILLABLE_RECV the notified accept/ - // getpeername could resume and repurpose these buffers between validation - // and the broker write. Fail closed before reading or writing anything, so - // this address-writing path is inert in legacy mode. Callers that pass a - // null address argument (accept with a null peer address) never reach here. + // workload memory: without WAIT_KILLABLE_RECV the notified getpeername + // could resume and repurpose these buffers between validation and the + // broker write. Fail closed before reading or writing anything, so this + // address-writing path is inert in legacy mode. if listener.writes_disabled() { return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); } @@ -1847,6 +1645,7 @@ fn error_to_errno(error: &io::Error) -> i32 { mod tests { use super::*; use openshell_isolation_interface::linux::seccomp_notify::ListenerMode; + use openshell_isolation_interface::linux::socket_confinement; #[test] fn provider_files_are_opened_on_demand_and_replaced() { @@ -2026,11 +1825,22 @@ mod tests { ))); } + fn duplicate_close_on_exec(fd: RawFd) -> io::Result { + // SAFETY: F_DUPFD_CLOEXEC returns an independent owned descriptor for + // the same open-file description. + let duplicate = unsafe { libc::fcntl(fd, libc::F_DUPFD_CLOEXEC, 3) }; + if duplicate < 0 { + return Err(io::Error::last_os_error()); + } + // SAFETY: successful fcntl returned one newly owned descriptor. + Ok(unsafe { OwnedFd::from_raw_fd(duplicate) }) + } + #[test] fn legacy_listener_rejects_socket_addr_write() { - // accept-with-address and getpeername both route through - // write_socket_addr; on a LegacyReadOnly listener the path must fail - // closed (EOPNOTSUPP) before any task-memory access. + // Relayed getpeername routes through write_socket_addr; on a + // LegacyReadOnly listener the path must fail closed (EOPNOTSUPP) + // before any task-memory access. // SAFETY: dup returns a new descriptor or a negative error. let dup = unsafe { libc::dup(libc::STDERR_FILENO) }; assert!(dup >= 0, "dup stderr"); @@ -2383,42 +2193,49 @@ mod tests { } #[test] - fn accepted_loopback_stream_is_registered_for_notified_operations() { + fn native_accept_inherits_loopback_confinement() { let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() .expect("start workload launcher"); let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); let (ready_tx, ready_rx) = std::sync::mpsc::sync_channel(1); let workload = std::thread::spawn(move || { launcher - .execute(move || -> io::Result { - let listener = TcpListener::bind("127.0.0.1:0")?; - ready_tx - .send(listener.local_addr()?) - .map_err(|_| io::Error::other("test client disappeared"))?; - let (stream, _) = listener.accept()?; - let peer = stream.peer_addr()?; - let payload = b"accepted"; - let iov = libc::iovec { - iov_base: payload.as_ptr().cast_mut().cast(), - iov_len: payload.len(), - }; - let message = libc::msghdr { - msg_name: std::ptr::null_mut(), - msg_namelen: 0, - msg_iov: (&raw const iov).cast_mut(), - msg_iovlen: 1, - msg_control: std::ptr::null_mut(), - msg_controllen: 0, - msg_flags: 0, - }; - // SAFETY: message references one live immutable payload; - // the accepted stream remains open for the call. - let sent = unsafe { libc::sendmsg(stream.as_raw_fd(), &raw const message, 0) }; - if sent != isize::try_from(payload.len()).expect("payload fits isize") { - return Err(io::Error::last_os_error()); - } - Ok(peer) - }) + .execute( + move || -> io::Result<(SocketAddr, SocketAddr, Option>)> { + let listener = TcpListener::bind("127.0.0.1:0")?; + ready_tx + .send(listener.local_addr()?) + .map_err(|_| io::Error::other("test client disappeared"))?; + // std passes a peer-address buffer, which the broker + // could not fill on a legacy listener. Native accept + // reports it directly from the kernel. + let (stream, accepted_peer) = listener.accept()?; + let peer = stream.peer_addr()?; + let device = socket_confinement::bound_device(stream.as_raw_fd())?; + let payload = b"accepted"; + let iov = libc::iovec { + iov_base: payload.as_ptr().cast_mut().cast(), + iov_len: payload.len(), + }; + let message = libc::msghdr { + msg_name: std::ptr::null_mut(), + msg_namelen: 0, + msg_iov: (&raw const iov).cast_mut(), + msg_iovlen: 1, + msg_control: std::ptr::null_mut(), + msg_controllen: 0, + msg_flags: 0, + }; + // SAFETY: message references one live immutable payload; + // the accepted stream remains open for the call. + let sent = + unsafe { libc::sendmsg(stream.as_raw_fd(), &raw const message, 0) }; + if sent != isize::try_from(payload.len()).expect("payload fits isize") { + return Err(io::Error::last_os_error()); + } + Ok((accepted_peer, peer, device)) + }, + ) .expect("launcher result") }); @@ -2435,14 +2252,298 @@ mod tests { .read_exact(&mut payload) .expect("read accepted stream"); assert_eq!(&payload, b"accepted"); - assert!( - workload - .join() - .expect("join workload") - .expect("accepted workload") - .ip() - .is_loopback() + let (accepted_peer, peer, device) = workload + .join() + .expect("join workload") + .expect("accepted workload"); + let client_address = client.local_addr().unwrap(); + assert_eq!(accepted_peer, client_address); + assert_eq!(peer, client_address); + assert_eq!(device.as_deref(), Some(&b"lo"[..])); + } + + /// Bound device name and the errno from an attempted rebind. + type ConfinementObservation = (Option>, Option); + + #[test] + fn direct_syscall_accept_and_getpeername_report_native_peer() { + // Static binaries and Go issue raw syscalls without a libc wrapper. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let (ready_tx, ready_rx) = std::sync::mpsc::sync_channel(1); + let workload = std::thread::spawn(move || { + launcher + .execute(move || -> io::Result<(SocketAddr, SocketAddr)> { + let listener = TcpListener::bind("127.0.0.1:0")?; + ready_tx + .send(listener.local_addr()?) + .map_err(|_| io::Error::other("test client disappeared"))?; + let mut storage = std::mem::MaybeUninit::::zeroed(); + let mut length = libc::socklen_t::try_from(size_of::()) + .expect("sockaddr_storage fits socklen_t"); + // SAFETY: storage and length are live, writable outputs. + let accepted = unsafe { + libc::syscall( + libc::SYS_accept4, + listener.as_raw_fd(), + storage.as_mut_ptr(), + &raw mut length, + libc::SOCK_CLOEXEC, + ) + }; + if accepted < 0 { + return Err(io::Error::last_os_error()); + } + let accepted = RawFd::try_from(accepted).map_err(io::Error::other)?; + // SAFETY: successful accept4 returned one owned descriptor. + let accepted = unsafe { OwnedFd::from_raw_fd(accepted) }; + // SAFETY: accept4 initialized the reported address prefix. + let accepted_peer = decode_sockaddr( + unsafe { storage.assume_init() }, + usize::try_from(length).unwrap_or(0), + )?; + let mut storage = std::mem::MaybeUninit::::zeroed(); + let mut length = libc::socklen_t::try_from(size_of::()) + .expect("sockaddr_storage fits socklen_t"); + // SAFETY: storage and length are live, writable outputs. + if unsafe { + libc::syscall( + libc::SYS_getpeername, + accepted.as_raw_fd(), + storage.as_mut_ptr(), + &raw mut length, + ) + } < 0 + { + return Err(io::Error::last_os_error()); + } + // SAFETY: getpeername initialized the reported prefix. + let peer = decode_sockaddr( + unsafe { storage.assume_init() }, + usize::try_from(length).unwrap_or(0), + )?; + Ok((accepted_peer, peer)) + }) + .expect("launcher result") + }); + let address = ready_rx.recv().expect("workload listener ready"); + let client = TcpStream::connect(address).expect("connect loopback client"); + let (accepted_peer, peer) = workload + .join() + .expect("join workload") + .expect("direct-syscall accept"); + assert_eq!(accepted_peer, client.local_addr().unwrap()); + assert_eq!(peer, accepted_peer); + } + + #[test] + #[ignore = "requires OPENSHELL_STATIC_SERVER pointing at a static test server"] + fn static_server_accepts_under_workload_filter() { + // The server listens on argv[1], prints "ready", accepts one + // connection, and writes the peer address it observed. + use std::io::BufRead as _; + let server = std::env::var("OPENSHELL_STATIC_SERVER").expect("OPENSHELL_STATIC_SERVER"); + let address = { + let probe = TcpListener::bind("127.0.0.1:0").unwrap(); + probe.local_addr().unwrap() + }; + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let connections: usize = std::env::var("OPENSHELL_STATIC_SERVER_CONNECTIONS") + .map_or(1, |value| value.parse().expect("connection count")); + let mut child = launcher + .execute(move || { + std::process::Command::new(server) + .arg(address.to_string()) + .arg(connections.to_string()) + .stdout(std::process::Stdio::piped()) + .stderr(std::process::Stdio::piped()) + .spawn() + }) + .unwrap() + .expect("spawn static server under the workload filter"); + let mut stdout = io::BufReader::new(child.stdout.take().unwrap()); + let mut line = String::new(); + stdout.read_line(&mut line).unwrap(); + assert_eq!(line.trim(), "ready", "server did not start"); + // Earlier connections measure accept latency through the workload + // filter; the last one also verifies the observed peer below. + let mut latencies = Vec::with_capacity(connections); + let started = Instant::now(); + for _ in 1..connections { + let begin = Instant::now(); + let mut client = TcpStream::connect(address).expect("connect to static server"); + let mut reply = String::new(); + client.read_to_string(&mut reply).unwrap(); + latencies.push(begin.elapsed()); + } + let begin = Instant::now(); + let mut client = TcpStream::connect(address).expect("connect to static server"); + let mut reply = String::new(); + client.read_to_string(&mut reply).unwrap(); + latencies.push(begin.elapsed()); + let total = started.elapsed(); + latencies.sort_unstable(); + let percentile = |p: usize| latencies[(latencies.len() - 1) * p / 100]; + eprintln!( + "static server: {connections} connections in {total:?}; p50 {:?} p99 {:?} max {:?}", + percentile(50), + percentile(99), + latencies[latencies.len() - 1] ); + let status = child.wait().unwrap(); + let mut stderr = String::new(); + child + .stderr + .take() + .unwrap() + .read_to_string(&mut stderr) + .unwrap(); + assert!(status.success(), "static server failed: {stderr}"); + assert_eq!( + reply.trim(), + client.local_addr().unwrap().to_string(), + "server observed the wrong peer" + ); + } + + #[test] + fn workload_sockets_are_bound_to_loopback_and_cannot_be_rebound() { + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let results = launcher + .execute(|| -> io::Result> { + let mut results = Vec::new(); + for (domain, kind) in [ + (libc::AF_INET, libc::SOCK_STREAM), + (libc::AF_INET, libc::SOCK_DGRAM), + (libc::AF_INET6, libc::SOCK_STREAM), + ] { + // SAFETY: scalar socket arguments; success returns one fd. + let fd = unsafe { libc::socket(domain, kind | libc::SOCK_CLOEXEC, 0) }; + if fd < 0 { + return Err(io::Error::last_os_error()); + } + // SAFETY: successful socket returned one owned descriptor. + let socket = unsafe { OwnedFd::from_raw_fd(fd) }; + let device = socket_confinement::bound_device(socket.as_raw_fd())?; + let name = c"eth0".to_bytes_with_nul(); + // SAFETY: name is a live NUL-terminated buffer. + let rebind = unsafe { + libc::setsockopt( + socket.as_raw_fd(), + libc::SOL_SOCKET, + libc::SO_BINDTODEVICE, + name.as_ptr().cast(), + libc::socklen_t::try_from(name.len()).expect("name fits socklen_t"), + ) + }; + let rebind_error = (rebind < 0) + .then(|| io::Error::last_os_error().raw_os_error()) + .flatten(); + results.push((device, rebind_error)); + } + Ok(results) + }) + .expect("launcher result") + .expect("workload sockets"); + for (device, rebind_error) in results { + assert_eq!(device.as_deref(), Some(&b"lo"[..])); + assert_eq!(rebind_error, Some(libc::EPERM)); + } + } + + #[test] + fn fast_open_sends_are_denied_for_every_descriptor() { + let local = TcpListener::bind("127.0.0.1:0").unwrap(); + local.set_nonblocking(true).unwrap(); + let address = local.local_addr().unwrap(); + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let error = launcher + .execute(move || -> io::Result<()> { + // SAFETY: scalar socket arguments; success returns one fd. + let fd = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0) }; + if fd < 0 { + return Err(io::Error::last_os_error()); + } + // SAFETY: successful socket returned one owned descriptor. + let socket = unsafe { OwnedFd::from_raw_fd(fd) }; + with_sockaddr(address, |native, length| { + // SAFETY: payload and native address are live for the call. + let sent = unsafe { + libc::sendto( + socket.as_raw_fd(), + b"x".as_ptr().cast(), + 1, + libc::MSG_FASTOPEN, + native, + length, + ) + }; + if sent < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(()) + } + }) + }) + .unwrap() + .unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EPERM)); + assert_eq!( + local.accept().unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + } + + #[test] + fn send_flags_read_the_scalar_argument_for_each_syscall() { + let flags = u64::try_from(libc::MSG_FASTOPEN).unwrap(); + assert_eq!( + send_flags(libc::SYS_sendto, [0, 0, 0, flags, 0, 0]), + libc::MSG_FASTOPEN + ); + assert_eq!( + send_flags(libc::SYS_sendmsg, [0, 0, flags, 0, 0, 0]), + libc::MSG_FASTOPEN + ); + assert_eq!( + send_flags(libc::SYS_sendmmsg, [0, 0, 0, flags | (1 << 32), 0, 0]), + libc::MSG_FASTOPEN + ); + } + + #[test] + fn interface_selection_options_are_denied() { + for (level, option) in [ + (libc::SOL_SOCKET, libc::SO_BINDTODEVICE), + (libc::SOL_SOCKET, libc::SO_BINDTOIFINDEX), + (libc::IPPROTO_IP, libc::IP_UNICAST_IF), + (libc::IPPROTO_IP, libc::IP_MULTICAST_IF), + (libc::IPPROTO_IPV6, libc::IPV6_UNICAST_IF), + (libc::IPPROTO_IPV6, libc::IPV6_MULTICAST_IF), + (libc::IPPROTO_IPV6, libc::IPV6_ADDRFORM), + (libc::IPPROTO_TCP, libc::TCP_FASTOPEN_CONNECT), + ] { + assert!(socket_option_is_denied(level, option), "{level}/{option}"); + } + assert!(!socket_option_is_denied( + libc::SOL_SOCKET, + libc::SO_REUSEADDR + )); + assert!(!socket_option_is_denied( + libc::IPPROTO_TCP, + libc::TCP_NODELAY + )); + assert!(!socket_option_is_denied( + libc::IPPROTO_IPV6, + libc::IPV6_V6ONLY + )); } #[test] diff --git a/docs/about/support-matrix.mdx b/docs/about/support-matrix.mdx index 4115854e8e..326165aeff 100644 --- a/docs/about/support-matrix.mdx +++ b/docs/about/support-matrix.mdx @@ -171,6 +171,7 @@ when it runs inside a container or microVM: | [Landlock LSM](https://docs.kernel.org/security/landlock.html) | Required | ABI 3 or newer, introduced in Linux 6.2, with Landlock enabled. The mandatory baseline protects private channel and bootstrap files, including against truncation. A filesystem policy's `best_effort` setting never disables this baseline. | | seccomp | Required | Nested user-notification filters and atomic `SECCOMP_IOCTL_NOTIF_ADDFD` with `SECCOMP_ADDFD_FLAG_SEND`, usable under the runtime's existing seccomp profile without added capabilities. The sandbox actively probes these operations before admitting the workload. | | Task-memory access | Required | The non-dumpable broker must be able to read and write a same-UID, dumpable workload child's memory through `process_vm_readv` / `process_vm_writev` or `/proc//mem`. The sandbox actively probes the production parent-to-child topology before admitting the workload. | +| Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE` before handing it to the workload, without added capabilities (Linux 5.14+). Accepted sockets inherit the binding, so `accept` runs natively and the number of accepted connections is limited only by the workload's open-file limit (`RLIMIT_NOFILE`). The sandbox actively probes that the binding installs, cannot be cleared, and is inherited before admitting the workload. | | seccomp `WAIT_KILLABLE_RECV` | Recommended (Linux 5.19+) | Keeps a notified workload thread in a kill-only wait so the broker can safely write mediated results into workload memory. Without it (kernels < 5.19, for example RHEL 9.x / RHCOS 5.14) the sandbox still starts, in a reduced **legacy read-only** mode described below. | A kernel version alone does not establish support. A disabled Landlock LSM or a @@ -193,16 +194,14 @@ DNS and TCP authorization); the only difference is that the broker refuses the mediated operations that write results back into workload memory, failing them closed with `EOPNOTSUPP`: -- `getpeername`; -- `accept` / `accept4` **when a non-null peer-address argument is supplied** - (a null address argument still works); +- `getpeername` on connections relayed through the supervisor, which would + report the original upstream address; - `sendmmsg` paths that write per-message lengths back to the caller. -Socket creation, `connect`, `bind`, `listen`, `sendto`, and `sendmsg` are -unaffected. Outbound-oriented workloads generally run unchanged; server -workloads whose accept wrappers request the peer address will see `EOPNOTSUPP` -until the node runs a kernel that provides `WAIT_KILLABLE_RECV` (Linux 5.19+, or -a distribution backport). The selected mode is reported in the sandbox +Socket creation, `connect`, `bind`, `listen`, `accept`, `sendto`, `sendmsg`, +and `getpeername` on local connections are unaffected. Local server workloads, +including those that read the peer address on accept, run unchanged. The +selected mode is reported in the sandbox qualification output as `seccomp_listener_mode` (`killable` or `legacy_read_only`). diff --git a/docs/kubernetes/openshift.mdx b/docs/kubernetes/openshift.mdx index c6189b1ebb..1c048e2628 100644 --- a/docs/kubernetes/openshift.mdx +++ b/docs/kubernetes/openshift.mdx @@ -21,15 +21,18 @@ OpenShell fails sandbox startup when either capability-free runtime probe fails. ## Node kernel and legacy read-only mode +OpenShell requires OpenShift 4.19 or later. The RHCOS kernels in OpenShift 4.16 +through 4.18 (RHEL 9.4, 5.14.0-427) are built without Landlock, so sandbox +startup fails its Landlock probe on those releases. + OpenShift nodes run RHCOS, which currently ships a RHEL 9.x kernel (5.14). That kernel predates `SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` (Linux 5.19), so the sandbox starts in a reduced **legacy read-only** cancellation mode. Isolation is unchanged, but the broker fails closed with `EOPNOTSUPP` on the mediated -operations that write results back into workload memory — `getpeername`, -`accept`/`accept4` with a non-null peer-address argument, and `sendmmsg` -per-message length write-backs. Outbound-oriented workloads run unchanged; -server workloads that read the peer address on accept need a node kernel with -`WAIT_KILLABLE_RECV` (Linux 5.19+, or a distribution backport). See the +operations that write results back into workload memory: `getpeername` on +connections relayed through the supervisor, and `sendmmsg` per-message length +write-backs. Local servers, including `accept` with a peer address, run +unchanged. See the [support matrix](/about/support-matrix#legacy-read-only-mode-kernels-before-linux-519) for the full behavior; the selected mode is reported as `seccomp_listener_mode` in the sandbox qualification output. From dc6449fd6efbbdb8e246cfd362d4dfb1a04158ff Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 17:32:40 -0700 Subject: [PATCH 02/18] fix(sandbox): close review gaps in native local accept confinement - Reject a TCP boundary control listener on a loopback address, including IPv4-mapped loopback. Production drivers bind the unspecified address; a loopback listener would be reachable from workload sockets the broker does not track. - Extend the confinement probe to IPv6 and prove that an accepted socket keeps its loopback binding after an AF_UNSPEC disconnect. - Pin the accepted-peer contract with tests: a loopback-bound listener admits only clients in the sandbox network namespace, and workload sockets cannot bind a non-loopback source address. - Correct the support matrix: accepted connections are bounded by per-process descriptor limits, the runtime PID limit, and the sandbox memory limit, not by a broker limit. Signed-off-by: Drew Newberry --- .../src/linux/socket_confinement.rs | 136 +++++++++++++++--- .../openshell-sandbox/src/boundary_server.rs | 52 +++++++ .../openshell-sandbox/src/network_broker.rs | 20 +++ docs/about/support-matrix.mdx | 2 +- 4 files changed, 193 insertions(+), 17 deletions(-) diff --git a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs index faa8a5ba37..7f2c0ef546 100644 --- a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs +++ b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs @@ -129,10 +129,13 @@ fn reject_ingress_interface(fd: RawFd, index: u32) -> io::Result<()> { /// Actively prove loopback confinement under the current runtime profile. /// -/// For each supported workload socket type this installs the binding, proves -/// that the sandbox credentials cannot clear or replace it, and proves that a -/// stream accepted from a confined listener inherits it. IPv6 is skipped only -/// when the kernel does not provide the address family. +/// For each supported workload socket type this installs the binding and +/// proves that the sandbox credentials cannot clear or replace it. For IPv4 +/// and IPv6 it proves that a stream accepted from a confined listener inherits +/// the binding and keeps it after an `AF_UNSPEC` disconnect. IPv6 is skipped +/// only when the kernel or namespace does not provide it. Routed ingress and +/// egress cannot be exercised without a non-loopback route, so those +/// guarantees rest on the kernel behavior verified per target kernel. /// /// # Errors /// @@ -189,22 +192,70 @@ fn probe_binding_is_immutable(fd: RawFd) -> io::Result<()> { } fn probe_accept_inherits_binding() -> io::Result<()> { - let listener = std::net::TcpListener::bind((std::net::Ipv4Addr::LOCALHOST, 0))?; - confine_to_loopback(listener.as_raw_fd()) - .map_err(|error| probe_error("confine probe listener", &error))?; - let client = std::net::TcpStream::connect(listener.local_addr()?)?; - let (accepted, peer) = listener.accept()?; - if peer != client.local_addr()? { - return Err(io::Error::other("accepted probe peer mismatch")); - } - if bound_device(accepted.as_raw_fd())?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { - return Err(io::Error::other( - "accepted socket did not inherit the loopback binding", - )); + for loopback in [ + std::net::IpAddr::V4(std::net::Ipv4Addr::LOCALHOST), + std::net::IpAddr::V6(std::net::Ipv6Addr::LOCALHOST), + ] { + let listener = match std::net::TcpListener::bind((loopback, 0)) { + Ok(listener) => listener, + // Kernels or namespaces without IPv6 have no ::1 to bind. + Err(error) + if loopback.is_ipv6() + && matches!( + error.raw_os_error(), + Some(libc::EAFNOSUPPORT | libc::EADDRNOTAVAIL) + ) => + { + continue; + } + Err(error) => return Err(error), + }; + confine_to_loopback(listener.as_raw_fd()) + .map_err(|error| probe_error("confine probe listener", &error))?; + let client = std::net::TcpStream::connect(listener.local_addr()?)?; + let (accepted, peer) = listener.accept()?; + if peer != client.local_addr()? { + return Err(io::Error::other("accepted probe peer mismatch")); + } + if bound_device(accepted.as_raw_fd())?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + return Err(io::Error::other( + "accepted socket did not inherit the loopback binding", + )); + } + // Natively accepted sockets are not tracked by the broker, so a + // workload can disconnect and reconnect them. The binding must + // survive that transition. + disconnect(accepted.as_raw_fd()) + .map_err(|error| probe_error("disconnect accepted probe socket", &error))?; + if bound_device(accepted.as_raw_fd())?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + return Err(io::Error::other( + "accepted socket lost the loopback binding after disconnect", + )); + } } Ok(()) } +fn disconnect(fd: RawFd) -> io::Result<()> { + // SAFETY: zeroed sockaddr storage with AF_UNSPEC is the documented + // disconnect request; the buffer is live for the call. + let mut address: libc::sockaddr_storage = unsafe { std::mem::zeroed() }; + address.ss_family = libc::sa_family_t::try_from(libc::AF_UNSPEC).map_err(io::Error::other)?; + // SAFETY: `address` is a live sockaddr_storage of the given length. + let result = unsafe { + libc::connect( + fd, + (&raw const address).cast(), + socklen(size_of::())?, + ) + }; + if result < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(()) + } +} + fn probe_error(context: &str, error: &io::Error) -> io::Error { io::Error::new(error.kind(), format!("{context}: {error}")) } @@ -319,6 +370,59 @@ mod tests { assert_eq!(&buffer, b"ping"); } + /// Return a local non-loopback address, if this namespace has one. + fn local_non_loopback_address() -> Option { + let probe = std::net::UdpSocket::bind((Ipv4Addr::UNSPECIFIED, 0)).ok()?; + probe.connect((Ipv4Addr::new(192, 0, 2, 1), 9)).ok()?; + match probe.local_addr().ok()?.ip() { + std::net::IpAddr::V4(address) if !address.is_loopback() => Some(address), + _ => None, + } + } + + #[test] + fn confined_listener_peers_are_limited_to_this_network_namespace() { + // Packets that arrive on another interface never match a listener + // bound to loopback. The only non-loopback peer address it can see is + // a client in the same namespace that binds its source to a local + // non-loopback address and connects to loopback; that client could + // equally connect from 127.0.0.1. Workload sockets cannot bind such a + // source, because the broker only permits loopback or unspecified + // binds. + let Some(local) = local_non_loopback_address() else { + eprintln!("skipping: no non-loopback IPv4 address in this namespace"); + return; + }; + let listener = TcpListener::bind((Ipv4Addr::UNSPECIFIED, 0)).unwrap(); + confine_to_loopback(listener.as_raw_fd()).unwrap(); + listener.set_nonblocking(true).unwrap(); + let port = listener.local_addr().unwrap().port(); + + let connect_from = |source: Ipv4Addr, destination: Ipv4Addr| { + let client = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None).unwrap(); + client.bind(&SocketAddr::from((source, 0)).into()).unwrap(); + client + .connect_timeout( + &SocketAddr::from((destination, port)).into(), + Duration::from_millis(300), + ) + .map(|()| client) + }; + + // The host's own address is matched against its real interface. + assert!(connect_from(local, local).is_err()); + assert_eq!( + listener.accept().unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + + let client = connect_from(local, Ipv4Addr::LOCALHOST).expect("same-namespace client"); + let (_, peer) = listener.accept().unwrap(); + assert_eq!(peer, client.local_addr().unwrap().as_socket().unwrap()); + assert_eq!(peer.ip(), std::net::IpAddr::V4(local)); + } + #[test] fn ingress_filter_is_locked() { let listener = TcpListener::bind((Ipv4Addr::LOCALHOST, 0)).unwrap(); diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index e271fd4961..a115c8a8ae 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -342,6 +342,18 @@ mod linux { .to_string(), ); } + // The TCP control listener rejects loopback-interface ingress + // only when it serves a supervisor in another network namespace. + // A loopback listener would be reachable from workload sockets + // that the broker does not track, such as natively accepted ones. + BoundaryListenerConfig::TlsTcp { address, .. } + if address.ip().to_canonical().is_loopback() => + { + return Err( + "boundary TLS listener must not bind a loopback address; workloads share the loopback interface" + .to_string(), + ); + } BoundaryListenerConfig::Vsock { control_port: 0, .. } => { @@ -4674,6 +4686,46 @@ mod linux { validate_running_identity(&config.workload_identity, false).unwrap(); } + #[test] + fn tcp_control_listener_rejects_loopback_addresses() { + let directory = tempfile::tempdir().expect("temporary directory"); + let (server_tls, _client_tls) = stage_test_tls(directory.path(), "validate"); + let config = |address: &str| BoundaryConfig { + boundary_id: "sandbox-1".to_string(), + generation: "generation-1".to_string(), + session_id: test_session_id(), + session_rotation: openshell_core::jwt::SessionRotation::new(1) + .expect("session rotation"), + auth_epoch: CredentialEpoch::new(1).expect("auth epoch"), + gateway_id: "test-gateway".to_string(), + verification_keys: vec![test_verification_key()], + listener: BoundaryListenerConfig::TlsTcp { + address: address.parse().expect("valid address"), + tls: server_tls.clone(), + }, + resource_claims: std::collections::BTreeMap::new(), + resource_claim_files: std::collections::BTreeMap::new(), + workload_identity: test_workload_identity(), + outer_fence: test_outer_fence(), + child_env: std::collections::HashMap::new(), + }; + for address in [ + "127.0.0.1:5500", + "127.0.0.2:5500", + "[::1]:5500", + "[::ffff:127.0.0.1]:5500", + ] { + assert!( + validate_config(&config(address)).is_err(), + "{address} must be rejected" + ); + } + for address in ["0.0.0.0:5500", "[::]:5500", "10.42.0.7:5500"] { + validate_config(&config(address)) + .unwrap_or_else(|error| panic!("{address} must be accepted: {error}")); + } + } + #[test] fn runtime_resource_claim_file_must_match_admitted_claim() { let directory = tempfile::tempdir().expect("temporary directory"); diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index c60e061a2a..3cc094627b 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -2518,6 +2518,26 @@ mod tests { ); } + #[test] + fn workload_cannot_bind_a_non_loopback_source_address() { + // Workload sockets can only present loopback source addresses. A + // non-loopback peer on a loopback-bound workload listener is therefore + // a non-workload process in the same network namespace. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let errors = launcher + .execute(|| { + ["192.0.2.10:0", "[2001:db8::10]:0"].map(|address| { + UdpSocket::bind(address) + .err() + .and_then(|error| error.raw_os_error()) + }) + }) + .expect("launcher result"); + assert_eq!(errors, [Some(libc::EACCES), Some(libc::EACCES)]); + } + #[test] fn interface_selection_options_are_denied() { for (level, option) in [ diff --git a/docs/about/support-matrix.mdx b/docs/about/support-matrix.mdx index 326165aeff..4c87744e49 100644 --- a/docs/about/support-matrix.mdx +++ b/docs/about/support-matrix.mdx @@ -171,7 +171,7 @@ when it runs inside a container or microVM: | [Landlock LSM](https://docs.kernel.org/security/landlock.html) | Required | ABI 3 or newer, introduced in Linux 6.2, with Landlock enabled. The mandatory baseline protects private channel and bootstrap files, including against truncation. A filesystem policy's `best_effort` setting never disables this baseline. | | seccomp | Required | Nested user-notification filters and atomic `SECCOMP_IOCTL_NOTIF_ADDFD` with `SECCOMP_ADDFD_FLAG_SEND`, usable under the runtime's existing seccomp profile without added capabilities. The sandbox actively probes these operations before admitting the workload. | | Task-memory access | Required | The non-dumpable broker must be able to read and write a same-UID, dumpable workload child's memory through `process_vm_readv` / `process_vm_writev` or `/proc//mem`. The sandbox actively probes the production parent-to-child topology before admitting the workload. | -| Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE` before handing it to the workload, without added capabilities (Linux 5.14+). Accepted sockets inherit the binding, so `accept` runs natively and the number of accepted connections is limited only by the workload's open-file limit (`RLIMIT_NOFILE`). The sandbox actively probes that the binding installs, cannot be cleared, and is inherited before admitting the workload. | +| Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE` before handing it to the workload, without added capabilities (Linux 5.14+). Accepted sockets inherit the binding, so `accept` runs natively and only admits clients in the sandbox's own network namespace. Such a client that is not a workload process may present a non-loopback source address. No broker limit applies to accepted connections; each process's open-file limit (`RLIMIT_NOFILE`), the runtime PID limit, and the sandbox memory limit bound them. The sandbox actively probes that the binding installs, cannot be cleared, and is inherited before admitting the workload. | | seccomp `WAIT_KILLABLE_RECV` | Recommended (Linux 5.19+) | Keeps a notified workload thread in a kill-only wait so the broker can safely write mediated results into workload memory. Without it (kernels < 5.19, for example RHEL 9.x / RHCOS 5.14) the sandbox still starts, in a reduced **legacy read-only** mode described below. | A kernel version alone does not establish support. A disabled Landlock LSM or a From f18403c186e0fddc220a133ebc9c5bbd48269c62 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 17:48:01 -0700 Subject: [PATCH 03/18] refactor(sandbox): replace unsafe socket calls with socket2 and rustix Socket confinement now uses socket2 for device binding and the ingress filter, and rustix for interface lookup and AF_UNSPEC disconnect, so the module contains no unsafe code. The control-listener filter is no longer locked: no safe API exposes SO_LOCK_FILTER, and the listener descriptor never leaves the trusted sandbox process, which marks every descriptor above stdio close-on-exec before running workload code. Tests added for native accept use rustix and socket2 instead of raw libc calls. rustix issues accept4 and getpeername as raw syscalls, so the direct-syscall test keeps its meaning. The close-on-exec sweep test runs in a re-executed test process instead of a forked child. The close_range(CLOSE_RANGE_CLOEXEC) syscall in the pre_exec hook remains the only unsafe added by this branch; neither rustix nor nix wraps it. Signed-off-by: Drew Newberry --- .../openshell-isolation-interface/Cargo.toml | 4 +- .../src/linux/child_seccomp.rs | 62 ++-- .../src/linux/socket_confinement.rs | 346 +++++------------- .../openshell-sandbox/src/boundary_server.rs | 2 +- .../openshell-sandbox/src/network_broker.rs | 171 ++------- 5 files changed, 170 insertions(+), 415 deletions(-) diff --git a/crates/openshell-isolation-interface/Cargo.toml b/crates/openshell-isolation-interface/Cargo.toml index b863f312a4..fe3e6f9551 100644 --- a/crates/openshell-isolation-interface/Cargo.toml +++ b/crates/openshell-isolation-interface/Cargo.toml @@ -22,10 +22,10 @@ tokio = { workspace = true } libc = "0.2" [target.'cfg(target_os = "linux")'.dependencies] -rustix = { workspace = true, features = ["fs", "process"] } +rustix = { workspace = true, features = ["fs", "net", "process"] } +socket2 = { workspace = true, features = ["all"] } [dev-dependencies] -socket2 = { workspace = true } tokio = { workspace = true } [lints] diff --git a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs index e709e755b7..e6ce09d9f6 100644 --- a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs +++ b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs @@ -384,33 +384,45 @@ mod tests { use super::*; #[test] - #[allow(unsafe_code)] fn inherited_sockets_are_marked_close_on_exec_but_stdio_is_not() { - // SAFETY: scalar socket arguments; deliberately inheritable. - let socket = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0) }; - assert!(socket > 2); - // SAFETY: the child performs only async-signal-safe syscalls and exits. - let pid = unsafe { libc::fork() }; - assert!(pid >= 0); - if pid == 0 { - let swept = mark_inherited_descriptors_close_on_exec().is_ok(); - // SAFETY: F_GETFD reads one descriptor flag word. - let socket_flags = unsafe { libc::fcntl(socket, libc::F_GETFD) }; - // SAFETY: as above, for stderr. - let stderr_flags = unsafe { libc::fcntl(libc::STDERR_FILENO, libc::F_GETFD) }; - let ok = swept - && socket_flags & libc::FD_CLOEXEC != 0 - && stderr_flags >= 0 - && stderr_flags & libc::FD_CLOEXEC == 0; - // SAFETY: terminate the forked child without running destructors. - unsafe { libc::_exit(i32::from(!ok)) }; + // The sweep changes every descriptor in the calling process, so run + // it in a fresh copy of this test binary rather than the harness. + const CHILD_MARKER: &str = "OPENSHELL_CLOEXEC_SWEEP_CHILD"; + if std::env::var_os(CHILD_MARKER).is_some() { + let socket = rustix::net::socket( + rustix::net::AddressFamily::INET, + rustix::net::SocketType::STREAM, + None, + ) + .expect("inheritable socket"); + assert!( + !rustix::io::fcntl_getfd(&socket) + .unwrap() + .contains(rustix::io::FdFlags::CLOEXEC) + ); + mark_inherited_descriptors_close_on_exec().expect("sweep descriptors"); + assert!( + rustix::io::fcntl_getfd(&socket) + .unwrap() + .contains(rustix::io::FdFlags::CLOEXEC) + ); + assert!( + !rustix::io::fcntl_getfd(io::stderr()) + .unwrap() + .contains(rustix::io::FdFlags::CLOEXEC) + ); + return; } - let mut status = 0; - // SAFETY: wait for the child created above. - assert_eq!(unsafe { libc::waitpid(pid, &raw mut status, 0) }, pid); - // SAFETY: close the test-owned socket. - unsafe { libc::close(socket) }; - assert!(libc::WIFEXITED(status) && libc::WEXITSTATUS(status) == 0); + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "linux::child_seccomp::tests::inherited_sockets_are_marked_close_on_exec_but_stdio_is_not", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + .expect("run isolated sweep test"); + assert!(status.success(), "isolated sweep test failed"); } #[test] diff --git a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs index 7f2c0ef546..294eec3939 100644 --- a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs +++ b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs @@ -12,14 +12,12 @@ //! binding requires `CAP_NET_RAW` in the network namespace's owning user //! namespace, which the capability-free sandbox and workload do not hold. -#![allow(unsafe_code)] - -use std::ffi::CStr; use std::io; -use std::mem::size_of; -use std::os::fd::{AsRawFd as _, FromRawFd as _, OwnedFd, RawFd}; +use std::os::fd::AsFd; + +use socket2::{Domain, SockFilter, SockRef, Socket, Type}; -const LOOPBACK_DEVICE: &CStr = c"lo"; +const LOOPBACK_DEVICE: &[u8] = b"lo"; /// Bind `fd` to the loopback device and verify the kernel recorded it. /// @@ -27,22 +25,10 @@ const LOOPBACK_DEVICE: &CStr = c"lo"; /// /// Returns the kernel error when the binding cannot be installed, or `EPERM` /// when the socket is already bound to another device. -pub fn confine_to_loopback(fd: RawFd) -> io::Result<()> { - let name = LOOPBACK_DEVICE.to_bytes_with_nul(); - // SAFETY: `name` is a live NUL-terminated buffer for the duration of the call. - let result = unsafe { - libc::setsockopt( - fd, - libc::SOL_SOCKET, - libc::SO_BINDTODEVICE, - name.as_ptr().cast(), - socklen(name.len())?, - ) - }; - if result < 0 { - return Err(io::Error::last_os_error()); - } - if bound_device(fd)?.as_deref() == Some(LOOPBACK_DEVICE.to_bytes()) { +pub fn confine_to_loopback(fd: impl AsFd) -> io::Result<()> { + let socket = SockRef::from(&fd); + socket.bind_device(Some(LOOPBACK_DEVICE))?; + if socket.device()?.as_deref() == Some(LOOPBACK_DEVICE) { Ok(()) } else { Err(io::Error::from_raw_os_error(libc::EPERM)) @@ -54,28 +40,8 @@ pub fn confine_to_loopback(fd: RawFd) -> io::Result<()> { /// # Errors /// /// Returns the kernel error from `getsockopt(SO_BINDTODEVICE)`. -pub fn bound_device(fd: RawFd) -> io::Result>> { - let mut name = [0_u8; libc::IFNAMSIZ]; - let mut length = socklen(name.len())?; - // SAFETY: `name` and `length` are live, writable outputs sized together. - let result = unsafe { - libc::getsockopt( - fd, - libc::SOL_SOCKET, - libc::SO_BINDTODEVICE, - name.as_mut_ptr().cast(), - &raw mut length, - ) - }; - if result < 0 { - return Err(io::Error::last_os_error()); - } - let length = usize::try_from(length).unwrap_or(0).min(name.len()); - let name = name[..length] - .split(|byte| *byte == 0) - .next() - .unwrap_or_default(); - Ok((!name.is_empty()).then(|| name.to_vec())) +pub fn bound_device(fd: impl AsFd) -> io::Result>> { + SockRef::from(&fd).device() } /// Drop TCP/UDP ingress that arrives on the loopback interface. @@ -83,48 +49,43 @@ pub fn bound_device(fd: RawFd) -> io::Result>> { /// Attach this to a trusted listener whose legitimate clients are never in the /// same network namespace. Matching the ingress interface rather than the /// source address also rejects connections to the host's own non-loopback -/// address, which the kernel delivers through loopback. The filter is locked -/// so later code cannot remove it accidentally. +/// address, which the kernel delivers through loopback. The filter is not +/// locked: the listener descriptor never leaves the trusted sandbox process, +/// which marks every descriptor above stdio close-on-exec before running +/// workload code. /// /// # Errors /// -/// Returns the kernel error when the filter cannot be attached or locked. -pub fn reject_loopback_ingress(fd: RawFd) -> io::Result<()> { - // SAFETY: LOOPBACK_DEVICE is a valid NUL-terminated interface name. - let index = unsafe { libc::if_nametoindex(LOOPBACK_DEVICE.as_ptr()) }; - if index == 0 { - return Err(io::Error::last_os_error()); - } +/// Returns the kernel error when the interface index cannot be resolved or +/// the filter cannot be attached. +pub fn reject_loopback_ingress(fd: impl AsFd) -> io::Result<()> { + let index = rustix::net::netdevice::name_to_index(&fd, "lo")?; reject_ingress_interface(fd, index) } -fn reject_ingress_interface(fd: RawFd, index: u32) -> io::Result<()> { +fn reject_ingress_interface(fd: impl AsFd, index: u32) -> io::Result<()> { // Ancillary loads use the documented negative offset encoding. let ifindex_offset = (libc::SKF_AD_OFF + libc::SKF_AD_IFINDEX).cast_unsigned(); - let mut program = [ - filter_stmt(libc::BPF_LD | libc::BPF_W | libc::BPF_ABS, ifindex_offset), - filter_jump(libc::BPF_JMP | libc::BPF_JEQ | libc::BPF_K, index, 0, 1), - filter_stmt(libc::BPF_RET | libc::BPF_K, 0), - filter_stmt(libc::BPF_RET | libc::BPF_K, u32::MAX), + let program = [ + filter( + libc::BPF_LD | libc::BPF_W | libc::BPF_ABS, + 0, + 0, + ifindex_offset, + ), + filter(libc::BPF_JMP | libc::BPF_JEQ | libc::BPF_K, 0, 1, index), + filter(libc::BPF_RET | libc::BPF_K, 0, 0, 0), + filter(libc::BPF_RET | libc::BPF_K, 0, 0, u32::MAX), ]; - let filter = libc::sock_fprog { - len: u16::try_from(program.len()).map_err(io::Error::other)?, - filter: program.as_mut_ptr(), - }; - // SAFETY: `filter` references `program`, which outlives the call. - let result = unsafe { - libc::setsockopt( - fd, - libc::SOL_SOCKET, - libc::SO_ATTACH_FILTER, - (&raw const filter).cast(), - socklen(size_of::())?, - ) - }; - if result < 0 { - return Err(io::Error::last_os_error()); - } - set_int_option(fd, libc::SOL_SOCKET, libc::SO_LOCK_FILTER, 1) + SockRef::from(&fd).attach_filter(&program) +} + +#[allow( + clippy::cast_possible_truncation, + reason = "classic BPF opcodes are 16-bit by definition" +)] +const fn filter(code: u32, jt: u8, jf: u8, k: u32) -> SockFilter { + SockFilter::new(code as u16, jt, jf, k) } /// Actively prove loopback confinement under the current runtime profile. @@ -142,50 +103,35 @@ fn reject_ingress_interface(fd: RawFd, index: u32) -> io::Result<()> { /// Returns an error describing the first failed property. pub fn probe_loopback_confinement() -> io::Result<()> { for (domain, kind) in [ - (libc::AF_INET, libc::SOCK_STREAM), - (libc::AF_INET, libc::SOCK_DGRAM), - (libc::AF_INET6, libc::SOCK_STREAM), - (libc::AF_INET6, libc::SOCK_DGRAM), + (Domain::IPV4, Type::STREAM), + (Domain::IPV4, Type::DGRAM), + (Domain::IPV6, Type::STREAM), + (Domain::IPV6, Type::DGRAM), ] { - let socket = match new_socket(domain, kind) { + let socket = match Socket::new(domain, kind, None) { Ok(socket) => socket, Err(error) - if domain == libc::AF_INET6 && error.raw_os_error() == Some(libc::EAFNOSUPPORT) => + if domain == Domain::IPV6 && error.raw_os_error() == Some(libc::EAFNOSUPPORT) => { continue; } Err(error) => return Err(error), }; - confine_to_loopback(socket.as_raw_fd()) + confine_to_loopback(&socket) .map_err(|error| probe_error("install loopback binding", &error))?; - probe_binding_is_immutable(socket.as_raw_fd())?; + probe_binding_is_immutable(&socket)?; } probe_accept_inherits_binding() } -fn probe_binding_is_immutable(fd: RawFd) -> io::Result<()> { - let empty = [0_u8; 1]; - // SAFETY: `empty` is a live one-byte buffer; an empty name requests unbind. - let clear = unsafe { - libc::setsockopt( - fd, - libc::SOL_SOCKET, - libc::SO_BINDTODEVICE, - empty.as_ptr().cast(), - socklen(empty.len())?, - ) - }; - if clear == 0 { +fn probe_binding_is_immutable(socket: &Socket) -> io::Result<()> { + // `None` requests an unbind; replacing the device takes the same path. + if socket.bind_device(None).is_ok() { return Err(io::Error::other( "sandbox credentials can clear a socket device binding", )); } - if set_int_option(fd, libc::SOL_SOCKET, libc::SO_BINDTOIFINDEX, 0).is_ok() { - return Err(io::Error::other( - "sandbox credentials can clear a socket interface-index binding", - )); - } - if bound_device(fd)?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + if socket.device()?.as_deref() != Some(LOOPBACK_DEVICE) { return Err(io::Error::other("socket device binding changed")); } Ok(()) @@ -210,14 +156,14 @@ fn probe_accept_inherits_binding() -> io::Result<()> { } Err(error) => return Err(error), }; - confine_to_loopback(listener.as_raw_fd()) + confine_to_loopback(&listener) .map_err(|error| probe_error("confine probe listener", &error))?; let client = std::net::TcpStream::connect(listener.local_addr()?)?; let (accepted, peer) = listener.accept()?; if peer != client.local_addr()? { return Err(io::Error::other("accepted probe peer mismatch")); } - if bound_device(accepted.as_raw_fd())?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + if bound_device(&accepted)?.as_deref() != Some(LOOPBACK_DEVICE) { return Err(io::Error::other( "accepted socket did not inherit the loopback binding", )); @@ -225,9 +171,9 @@ fn probe_accept_inherits_binding() -> io::Result<()> { // Natively accepted sockets are not tracked by the broker, so a // workload can disconnect and reconnect them. The binding must // survive that transition. - disconnect(accepted.as_raw_fd()) - .map_err(|error| probe_error("disconnect accepted probe socket", &error))?; - if bound_device(accepted.as_raw_fd())?.as_deref() != Some(LOOPBACK_DEVICE.to_bytes()) { + rustix::net::connect_unspec(&accepted) + .map_err(|error| probe_error("disconnect accepted probe socket", &error.into()))?; + if bound_device(&accepted)?.as_deref() != Some(LOOPBACK_DEVICE) { return Err(io::Error::other( "accepted socket lost the loopback binding after disconnect", )); @@ -236,79 +182,10 @@ fn probe_accept_inherits_binding() -> io::Result<()> { Ok(()) } -fn disconnect(fd: RawFd) -> io::Result<()> { - // SAFETY: zeroed sockaddr storage with AF_UNSPEC is the documented - // disconnect request; the buffer is live for the call. - let mut address: libc::sockaddr_storage = unsafe { std::mem::zeroed() }; - address.ss_family = libc::sa_family_t::try_from(libc::AF_UNSPEC).map_err(io::Error::other)?; - // SAFETY: `address` is a live sockaddr_storage of the given length. - let result = unsafe { - libc::connect( - fd, - (&raw const address).cast(), - socklen(size_of::())?, - ) - }; - if result < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(()) - } -} - fn probe_error(context: &str, error: &io::Error) -> io::Error { io::Error::new(error.kind(), format!("{context}: {error}")) } -fn new_socket(domain: i32, kind: i32) -> io::Result { - // SAFETY: scalar socket arguments; success returns one owned descriptor. - let fd = unsafe { libc::socket(domain, kind | libc::SOCK_CLOEXEC, 0) }; - if fd < 0 { - return Err(io::Error::last_os_error()); - } - // SAFETY: successful socket returned one newly owned descriptor. - Ok(unsafe { OwnedFd::from_raw_fd(fd) }) -} - -fn set_int_option(fd: RawFd, level: i32, option: i32, value: i32) -> io::Result<()> { - // SAFETY: `value` is a live int for the duration of the call. - let result = unsafe { - libc::setsockopt( - fd, - level, - option, - (&raw const value).cast(), - socklen(size_of::())?, - ) - }; - if result < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(()) - } -} - -fn socklen(length: usize) -> io::Result { - libc::socklen_t::try_from(length).map_err(io::Error::other) -} - -const fn filter_stmt(code: u32, k: u32) -> libc::sock_filter { - filter_jump(code, k, 0, 0) -} - -#[allow( - clippy::cast_possible_truncation, - reason = "classic BPF opcodes are 16-bit by definition" -)] -const fn filter_jump(code: u32, k: u32, jt: u8, jf: u8) -> libc::sock_filter { - libc::sock_filter { - code: code as u16, - jt, - jf, - k, - } -} - #[cfg(test)] mod tests { use super::*; @@ -316,6 +193,10 @@ mod tests { use std::net::{Ipv4Addr, SocketAddr, TcpListener, TcpStream}; use std::time::Duration; + fn new_socket(domain: Domain, kind: Type) -> Socket { + Socket::new(domain, kind, None).unwrap() + } + #[test] fn active_probe_passes_without_capabilities() { probe_loopback_confinement().expect("loopback confinement probe"); @@ -323,19 +204,16 @@ mod tests { #[test] fn unbound_socket_reports_no_device() { - let socket = new_socket(libc::AF_INET, libc::SOCK_STREAM).unwrap(); - assert_eq!(bound_device(socket.as_raw_fd()).unwrap(), None); + let socket = new_socket(Domain::IPV4, Type::STREAM); + assert_eq!(bound_device(&socket).unwrap(), None); } #[test] fn confined_socket_cannot_be_rebound() { - let socket = new_socket(libc::AF_INET, libc::SOCK_DGRAM).unwrap(); - confine_to_loopback(socket.as_raw_fd()).unwrap(); - assert!(confine_to_loopback(socket.as_raw_fd()).is_err()); - assert_eq!( - bound_device(socket.as_raw_fd()).unwrap().as_deref(), - Some(&b"lo"[..]) - ); + let socket = new_socket(Domain::IPV4, Type::DGRAM); + confine_to_loopback(&socket).unwrap(); + assert!(confine_to_loopback(&socket).is_err()); + assert_eq!(bound_device(&socket).unwrap().as_deref(), Some(&b"lo"[..])); } fn connect_with_timeout(address: SocketAddr) -> io::Result { @@ -345,7 +223,7 @@ mod tests { #[test] fn loopback_ingress_filter_rejects_loopback_connections() { let listener = TcpListener::bind((Ipv4Addr::UNSPECIFIED, 0)).unwrap(); - reject_loopback_ingress(listener.as_raw_fd()).unwrap(); + reject_loopback_ingress(&listener).unwrap(); listener.set_nonblocking(true).unwrap(); let port = listener.local_addr().unwrap().port(); // Dropped SYNs never complete the handshake. @@ -361,7 +239,7 @@ mod tests { // Positive control: the same program keyed to an absent interface // index must leave loopback traffic untouched. let listener = TcpListener::bind((Ipv4Addr::LOCALHOST, 0)).unwrap(); - reject_ingress_interface(listener.as_raw_fd(), u32::MAX).unwrap(); + reject_ingress_interface(&listener, u32::MAX).unwrap(); let mut client = connect_with_timeout(listener.local_addr().unwrap()).unwrap(); let (mut accepted, _) = listener.accept().unwrap(); client.write_all(b"ping").unwrap(); @@ -394,13 +272,12 @@ mod tests { return; }; let listener = TcpListener::bind((Ipv4Addr::UNSPECIFIED, 0)).unwrap(); - confine_to_loopback(listener.as_raw_fd()).unwrap(); + confine_to_loopback(&listener).unwrap(); listener.set_nonblocking(true).unwrap(); let port = listener.local_addr().unwrap().port(); let connect_from = |source: Ipv4Addr, destination: Ipv4Addr| { - let client = - socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None).unwrap(); + let client = new_socket(Domain::IPV4, Type::STREAM); client.bind(&SocketAddr::from((source, 0)).into()).unwrap(); client .connect_timeout( @@ -423,24 +300,6 @@ mod tests { assert_eq!(peer.ip(), std::net::IpAddr::V4(local)); } - #[test] - fn ingress_filter_is_locked() { - let listener = TcpListener::bind((Ipv4Addr::LOCALHOST, 0)).unwrap(); - reject_loopback_ingress(listener.as_raw_fd()).unwrap(); - // SAFETY: SO_DETACH_FILTER ignores its value argument. - let detach = unsafe { - let value = 0_i32; - libc::setsockopt( - listener.as_raw_fd(), - libc::SOL_SOCKET, - libc::SO_DETACH_FILTER, - (&raw const value).cast(), - socklen(size_of::()).unwrap(), - ) - }; - assert!(detach < 0); - } - /// Opt-in checks against a real non-loopback topology. /// /// A trusted harness creates a network namespace whose non-loopback @@ -490,36 +349,24 @@ mod tests { } } - fn confined(domain: i32, kind: i32) -> OwnedFd { - let socket = new_socket(domain, kind | libc::SOCK_NONBLOCK).unwrap(); - confine_to_loopback(socket.as_raw_fd()).unwrap(); + fn confined(domain: Domain, kind: Type) -> Socket { + let socket = new_socket(domain, kind.nonblocking()); + confine_to_loopback(&socket).unwrap(); socket } - fn attempt_egress(fd: RawFd, kind: i32, destination: SocketAddr) -> Option { - let native = socket2::SockAddr::from(destination); - // SAFETY: payload and the native address are live for each call. - let result = unsafe { - match kind { - libc::SOCK_DGRAM => libc::sendto( - fd, - b"probe".as_ptr().cast(), - 5, - 0, - native.as_ptr().cast(), - native.len(), - ), - _ => libc::sendto( - fd, - b"probe".as_ptr().cast(), - 5, - libc::MSG_FASTOPEN, - native.as_ptr().cast(), - native.len(), - ), - } + fn errno(result: io::Result) -> Option { + result.err().map(|error| error.raw_os_error().unwrap_or(0)) + } + + fn attempt_egress(socket: &Socket, kind: Type, destination: SocketAddr) -> Option { + // Streams use Fast Open so the send itself attempts a connect. + let flags = if kind == Type::DGRAM { + 0 + } else { + libc::MSG_FASTOPEN }; - (result < 0).then(|| io::Error::last_os_error().raw_os_error().unwrap_or(0)) + errno(socket.send_to_with_flags(b"probe", &destination.into(), flags)) } #[test] @@ -552,28 +399,19 @@ mod tests { let baseline = quiet_counter(&device); let mut outcomes = Vec::new(); for destination in &destinations { - let domain = match destination { - SocketAddr::V4(_) => libc::AF_INET, - SocketAddr::V6(_) => libc::AF_INET6, - }; - for kind in [libc::SOCK_DGRAM, libc::SOCK_STREAM] { + let domain = Domain::for_address(*destination); + for kind in [Type::DGRAM, Type::STREAM] { let socket = confined(domain, kind); outcomes.push(( destination, kind, "send", - attempt_egress(socket.as_raw_fd(), kind, *destination), + attempt_egress(&socket, kind, *destination), )); - if kind == libc::SOCK_STREAM { + if kind == Type::STREAM { let socket = confined(domain, kind); - let native = socket2::SockAddr::from(*destination); - // SAFETY: native address is live for the call. - let result = unsafe { - libc::connect(socket.as_raw_fd(), native.as_ptr().cast(), native.len()) - }; - let errno = (result < 0) - .then(|| io::Error::last_os_error().raw_os_error().unwrap_or(0)); - outcomes.push((destination, kind, "connect", errno)); + let result = socket.connect(&(*destination).into()); + outcomes.push((destination, kind, "connect", errno(result))); std::thread::sleep(Duration::from_millis(200)); } } @@ -597,13 +435,11 @@ mod tests { let control_port: u16 = env("OPENSHELL_TOPOLOGY_CONTROL_PORT").parse().unwrap(); let wait = Duration::from_secs(env("OPENSHELL_TOPOLOGY_WAIT_SECS").parse().unwrap()); let listen = |port: u16, confine: bool| { - let socket = - socket2::Socket::new(socket2::Domain::IPV6, socket2::Type::STREAM, None) - .unwrap(); + let socket = new_socket(Domain::IPV6, Type::STREAM); // Dual-stack wildcard covers IPv4 and IPv4-mapped peers too. socket.set_only_v6(false).unwrap(); if confine { - confine_to_loopback(socket.as_raw_fd()).unwrap(); + confine_to_loopback(&socket).unwrap(); } socket .bind(&SocketAddr::from((std::net::Ipv6Addr::UNSPECIFIED, port)).into()) diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index a115c8a8ae..effd97dac2 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -3259,7 +3259,7 @@ mod linux { socket.set_reuse_address(true)?; if !address.ip().is_loopback() { openshell_isolation_interface::linux::socket_confinement::reject_loopback_ingress( - socket.as_raw_fd(), + &socket, )?; } socket.bind(&address.into())?; diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 3cc094627b..4d9de5cee4 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -638,9 +638,7 @@ fn create_socket( // Confinement is standing kernel state that must exist before the workload // can observe the descriptor. Natively accepted children inherit it, so // local accept needs no per-connection broker inspection. - openshell_isolation_interface::linux::socket_confinement::confine_to_loopback( - source.as_raw_fd(), - )?; + openshell_isolation_interface::linux::socket_confinement::confine_to_loopback(&source)?; let metadata = SocketMetadata { family, kind, @@ -1825,17 +1823,6 @@ mod tests { ))); } - fn duplicate_close_on_exec(fd: RawFd) -> io::Result { - // SAFETY: F_DUPFD_CLOEXEC returns an independent owned descriptor for - // the same open-file description. - let duplicate = unsafe { libc::fcntl(fd, libc::F_DUPFD_CLOEXEC, 3) }; - if duplicate < 0 { - return Err(io::Error::last_os_error()); - } - // SAFETY: successful fcntl returned one newly owned descriptor. - Ok(unsafe { OwnedFd::from_raw_fd(duplicate) }) - } - #[test] fn legacy_listener_rejects_socket_addr_write() { // Relayed getpeername routes through write_socket_addr; on a @@ -1866,11 +1853,10 @@ mod tests { }; let mut registry = SocketRegistry::new(1, 2).unwrap(); let mut create = || { - // SAFETY: a successful socket call returns a new owned descriptor. - let fd = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0) }; - assert!(fd >= 0); - let socket = unsafe { OwnedFd::from_raw_fd(fd) }; - let installed = duplicate_close_on_exec(fd).unwrap(); + let socket = OwnedFd::from( + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None).unwrap(), + ); + let installed = rustix::io::fcntl_dupfd_cloexec(&socket, 3).unwrap(); let tentative = registry.stage(socket, metadata).unwrap(); let identity = registry.commit(tentative).unwrap(); (installed, identity) @@ -2211,27 +2197,18 @@ mod tests { // reports it directly from the kernel. let (stream, accepted_peer) = listener.accept()?; let peer = stream.peer_addr()?; - let device = socket_confinement::bound_device(stream.as_raw_fd())?; + let device = socket_confinement::bound_device(&stream)?; + // sendmsg with no destination is notified and must + // continue for an untracked accepted stream. let payload = b"accepted"; - let iov = libc::iovec { - iov_base: payload.as_ptr().cast_mut().cast(), - iov_len: payload.len(), - }; - let message = libc::msghdr { - msg_name: std::ptr::null_mut(), - msg_namelen: 0, - msg_iov: (&raw const iov).cast_mut(), - msg_iovlen: 1, - msg_control: std::ptr::null_mut(), - msg_controllen: 0, - msg_flags: 0, - }; - // SAFETY: message references one live immutable payload; - // the accepted stream remains open for the call. - let sent = - unsafe { libc::sendmsg(stream.as_raw_fd(), &raw const message, 0) }; - if sent != isize::try_from(payload.len()).expect("payload fits isize") { - return Err(io::Error::last_os_error()); + let sent = rustix::net::sendmsg( + &stream, + &[io::IoSlice::new(payload)], + &mut rustix::net::SendAncillaryBuffer::default(), + rustix::net::SendFlags::empty(), + )?; + if sent != payload.len() { + return Err(io::Error::from_raw_os_error(libc::EIO)); } Ok((accepted_peer, peer, device)) }, @@ -2279,50 +2256,16 @@ mod tests { ready_tx .send(listener.local_addr()?) .map_err(|_| io::Error::other("test client disappeared"))?; - let mut storage = std::mem::MaybeUninit::::zeroed(); - let mut length = libc::socklen_t::try_from(size_of::()) - .expect("sockaddr_storage fits socklen_t"); - // SAFETY: storage and length are live, writable outputs. - let accepted = unsafe { - libc::syscall( - libc::SYS_accept4, - listener.as_raw_fd(), - storage.as_mut_ptr(), - &raw mut length, - libc::SOCK_CLOEXEC, - ) + // rustix issues accept4 and getpeername as raw syscalls. + let (accepted, accepted_peer) = + rustix::net::acceptfrom_with(&listener, rustix::net::SocketFlags::CLOEXEC)?; + let as_socket_addr = |address: Option| { + address + .and_then(|address| SocketAddr::try_from(address).ok()) + .ok_or_else(|| io::Error::from_raw_os_error(libc::EAFNOSUPPORT)) }; - if accepted < 0 { - return Err(io::Error::last_os_error()); - } - let accepted = RawFd::try_from(accepted).map_err(io::Error::other)?; - // SAFETY: successful accept4 returned one owned descriptor. - let accepted = unsafe { OwnedFd::from_raw_fd(accepted) }; - // SAFETY: accept4 initialized the reported address prefix. - let accepted_peer = decode_sockaddr( - unsafe { storage.assume_init() }, - usize::try_from(length).unwrap_or(0), - )?; - let mut storage = std::mem::MaybeUninit::::zeroed(); - let mut length = libc::socklen_t::try_from(size_of::()) - .expect("sockaddr_storage fits socklen_t"); - // SAFETY: storage and length are live, writable outputs. - if unsafe { - libc::syscall( - libc::SYS_getpeername, - accepted.as_raw_fd(), - storage.as_mut_ptr(), - &raw mut length, - ) - } < 0 - { - return Err(io::Error::last_os_error()); - } - // SAFETY: getpeername initialized the reported prefix. - let peer = decode_sockaddr( - unsafe { storage.assume_init() }, - usize::try_from(length).unwrap_or(0), - )?; + let accepted_peer = as_socket_addr(accepted_peer)?; + let peer = as_socket_addr(rustix::net::getpeername(&accepted)?)?; Ok((accepted_peer, peer)) }) .expect("launcher result") @@ -2418,32 +2361,16 @@ mod tests { .execute(|| -> io::Result> { let mut results = Vec::new(); for (domain, kind) in [ - (libc::AF_INET, libc::SOCK_STREAM), - (libc::AF_INET, libc::SOCK_DGRAM), - (libc::AF_INET6, libc::SOCK_STREAM), + (socket2::Domain::IPV4, socket2::Type::STREAM), + (socket2::Domain::IPV4, socket2::Type::DGRAM), + (socket2::Domain::IPV6, socket2::Type::STREAM), ] { - // SAFETY: scalar socket arguments; success returns one fd. - let fd = unsafe { libc::socket(domain, kind | libc::SOCK_CLOEXEC, 0) }; - if fd < 0 { - return Err(io::Error::last_os_error()); - } - // SAFETY: successful socket returned one owned descriptor. - let socket = unsafe { OwnedFd::from_raw_fd(fd) }; - let device = socket_confinement::bound_device(socket.as_raw_fd())?; - let name = c"eth0".to_bytes_with_nul(); - // SAFETY: name is a live NUL-terminated buffer. - let rebind = unsafe { - libc::setsockopt( - socket.as_raw_fd(), - libc::SOL_SOCKET, - libc::SO_BINDTODEVICE, - name.as_ptr().cast(), - libc::socklen_t::try_from(name.len()).expect("name fits socklen_t"), - ) - }; - let rebind_error = (rebind < 0) - .then(|| io::Error::last_os_error().raw_os_error()) - .flatten(); + let socket = socket2::Socket::new(domain, kind, None)?; + let device = socket_confinement::bound_device(&socket)?; + let rebind_error = socket + .bind_device(Some(b"eth0")) + .err() + .and_then(|error| error.raw_os_error()); results.push((device, rebind_error)); } Ok(results) @@ -2466,31 +2393,11 @@ mod tests { let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); let error = launcher .execute(move || -> io::Result<()> { - // SAFETY: scalar socket arguments; success returns one fd. - let fd = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0) }; - if fd < 0 { - return Err(io::Error::last_os_error()); - } - // SAFETY: successful socket returned one owned descriptor. - let socket = unsafe { OwnedFd::from_raw_fd(fd) }; - with_sockaddr(address, |native, length| { - // SAFETY: payload and native address are live for the call. - let sent = unsafe { - libc::sendto( - socket.as_raw_fd(), - b"x".as_ptr().cast(), - 1, - libc::MSG_FASTOPEN, - native, - length, - ) - }; - if sent < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(()) - } - }) + let socket = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None)?; + socket + .send_to_with_flags(b"x", &address.into(), libc::MSG_FASTOPEN) + .map(drop) }) .unwrap() .unwrap_err(); From f7e91a32da56d138a42b7e4d0499bd552d20580c Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 18:13:49 -0700 Subject: [PATCH 04/18] fix(sandbox): allowlist workload socket families The workload seccomp filter denied only AF_PACKET, AF_BLUETOOTH, and AF_VSOCK, and the broker continued socket() for every non-INET family. Several protocol families, including AF_RXRPC, AF_SMC, and AF_KCM, carry traffic over kernel-owned sockets that the broker never creates and that are not bound to loopback. Allow only AF_UNIX, AF_NETLINK (still limited to NETLINK_ROUTE), and the brokered AF_INET/AF_INET6 families, and restrict socketpair(2) to AF_UNIX. Both decisions use the scalar domain argument. The broker independently refuses non-INET families other than AF_UNIX and AF_NETLINK with EAFNOSUPPORT. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/network_broker.rs | 39 +++++- .../src/sandbox/linux/seccomp.rs | 132 ++++++++++++++---- docs/security/best-practices.mdx | 2 +- 3 files changed, 147 insertions(+), 26 deletions(-) diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 4d9de5cee4..5be49d4c21 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -598,7 +598,13 @@ fn create_socket( let domain = i32::try_from(notification.args[0]) .map_err(|_| io::Error::from_raw_os_error(libc::EAFNOSUPPORT))?; if !matches!(domain, libc::AF_INET | libc::AF_INET6) { - return listener.respond_continue(notification.id); + // The workload filter already refuses other families. Repeat the + // decision here so a filter change cannot let a kernel transport + // socket bypass loopback confinement; the domain is a scalar argument. + if matches!(domain, libc::AF_UNIX | libc::AF_NETLINK) { + return listener.respond_continue(notification.id); + } + return Err(io::Error::from_raw_os_error(libc::EAFNOSUPPORT)); } let raw_kind = i32::try_from(notification.args[1]) .map_err(|_| io::Error::from_raw_os_error(libc::EPROTONOSUPPORT))?; @@ -2425,6 +2431,37 @@ mod tests { ); } + #[test] + fn broker_refuses_socket_families_it_cannot_confine() { + // Independent of the static workload filter: the broker continues + // only Unix and netlink sockets and creates INET sockets itself. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let results = launcher + .execute(|| { + [ + (libc::AF_UNIX, socket2::Type::STREAM), + (libc::AF_RXRPC, socket2::Type::DGRAM), + (libc::AF_ALG, socket2::Type::SEQPACKET), + ] + .map(|(domain, kind)| { + socket2::Socket::new(socket2::Domain::from(domain), kind, None) + .map(drop) + .map_err(|error| error.raw_os_error()) + }) + }) + .expect("launcher result"); + assert_eq!( + results, + [ + Ok(()), + Err(Some(libc::EAFNOSUPPORT)), + Err(Some(libc::EAFNOSUPPORT)) + ] + ); + } + #[test] fn workload_cannot_bind_a_non_loopback_source_address() { // Workload sockets can only present loopback source addresses. A diff --git a/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs b/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs index 81d0b1630f..e9fa55cc1b 100644 --- a/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs +++ b/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs @@ -5,7 +5,11 @@ //! //! The filter uses a default-allow policy with targeted blocks: //! -//! 1. **Socket domain blocks** -- prevent raw/kernel sockets that bypass the proxy +//! 1. **Socket domain allowlist** -- only `AF_UNIX`, `AF_NETLINK`, and (when +//! networking is enabled) the brokered `AF_INET`/`AF_INET6` families can be +//! created. Every other family is refused, because protocol families such as +//! `AF_RXRPC`, `AF_SMC`, and `AF_KCM` carry traffic over kernel-owned +//! sockets that the broker never creates or confines to loopback. //! 2. **Unconditional syscall blocks** -- block syscalls that enable sandbox escape //! (fileless exec, ptrace, BPF, cross-process memory access, `io_uring`, mount) //! 3. **Conditional syscall blocks** -- block dangerous flag combinations on otherwise @@ -184,23 +188,18 @@ fn apply_runtime_filters( fn build_filter_rules(allow_inet: bool) -> Result>> { let mut rules: BTreeMap> = BTreeMap::new(); - // --- Socket domain blocks --- - let mut blocked_domains = vec![ - libc::AF_PACKET, - libc::AF_BLUETOOTH, - libc::AF_VSOCK, - // AF_NETLINK is handled separately below: NETLINK_ROUTE (protocol 0) - // is allowed for getifaddrs(3); all other netlink protocols are blocked. - ]; - if !allow_inet { - blocked_domains.push(libc::AF_INET); - blocked_domains.push(libc::AF_INET6); - } - - for domain in blocked_domains { - debug!(domain, "Blocking socket domain via seccomp"); - add_socket_domain_rule(&mut rules, domain)?; + // --- Socket domain allowlist --- + // AF_NETLINK is narrowed further below: only NETLINK_ROUTE (protocol 0) + // is allowed, for getifaddrs(3). + let mut allowed_domains = vec![libc::AF_UNIX, libc::AF_NETLINK]; + if allow_inet { + allowed_domains.extend([libc::AF_INET, libc::AF_INET6]); } + debug!(?allowed_domains, "Restricting socket domains via seccomp"); + add_socket_domain_allowlist(&mut rules, libc::SYS_socket, &allowed_domains)?; + // socketpair(2) is only meaningful for AF_UNIX here; other families either + // reject it or create kernel transport sockets. + add_socket_domain_allowlist(&mut rules, libc::SYS_socketpair, &[libc::AF_UNIX])?; // Allow AF_NETLINK only for NETLINK_ROUTE (protocol 0). // @@ -296,14 +295,27 @@ fn build_filter_rules(allow_inet: bool) -> Result Ok(rules) } +/// Refuse `syscall` unless its domain argument is one of `allowed`. +/// +/// A seccomp rule matches only when all of its conditions hold, so one rule +/// with a `!=` condition per allowed domain matches exactly the domains +/// outside the allowlist. The domain is a scalar argument that another thread +/// cannot replace before the kernel reads it. #[allow(clippy::cast_sign_loss)] -fn add_socket_domain_rule(rules: &mut BTreeMap>, domain: i32) -> Result<()> { - let condition = - SeccompCondition::new(0, SeccompCmpArgLen::Dword, SeccompCmpOp::Eq, domain as u64) - .into_diagnostic()?; - - let rule = SeccompRule::new(vec![condition]).into_diagnostic()?; - rules.entry(libc::SYS_socket).or_default().push(rule); +fn add_socket_domain_allowlist( + rules: &mut BTreeMap>, + syscall: i64, + allowed: &[i32], +) -> Result<()> { + let conditions = allowed + .iter() + .map(|domain| { + SeccompCondition::new(0, SeccompCmpArgLen::Dword, SeccompCmpOp::Ne, *domain as u64) + .into_diagnostic() + }) + .collect::>>()?; + let rule = SeccompRule::new(conditions).into_diagnostic()?; + rules.entry(syscall).or_default().push(rule); Ok(()) } @@ -851,6 +863,78 @@ mod tests { ); } + #[test] + fn behavioral_socket_families_are_allowlisted() { + // Applying a filter is irreversible, so run the probe in a fresh copy + // of this test binary rather than in the harness process. + const CHILD_MARKER: &str = "OPENSHELL_SOCKET_FAMILY_ALLOWLIST_CHILD"; + // libc does not export these family numbers. + const AF_KCM: i32 = 41; + const AF_SMC: i32 = 43; + if std::env::var_os(CHILD_MARKER).is_some() { + set_no_new_privs().expect("set no_new_privs"); + apply_filter(&build_filter(true).unwrap()).expect("apply proxy-mode filter"); + let create = |domain: i32, kind: socket2::Type| { + socket2::Socket::new(socket2::Domain::from(domain), kind, None) + .map(drop) + .map_err(|error| error.raw_os_error()) + }; + for (domain, kind) in [ + (libc::AF_UNIX, socket2::Type::STREAM), + (libc::AF_NETLINK, socket2::Type::RAW), + (libc::AF_INET, socket2::Type::STREAM), + (libc::AF_INET6, socket2::Type::DGRAM), + ] { + assert_eq!( + create(domain, kind), + Ok(()), + "domain {domain} must be allowed" + ); + } + // Families whose kernel transport sockets the broker cannot + // confine, plus previously denylisted ones. + for (domain, kind) in [ + (libc::AF_RXRPC, socket2::Type::DGRAM), + (AF_SMC, socket2::Type::STREAM), + (AF_KCM, socket2::Type::DGRAM), + (libc::AF_ALG, socket2::Type::SEQPACKET), + (libc::AF_TIPC, socket2::Type::from(libc::SOCK_RDM)), + (libc::AF_PACKET, socket2::Type::RAW), + (libc::AF_VSOCK, socket2::Type::STREAM), + ] { + assert_eq!( + create(domain, kind), + Err(Some(libc::EPERM)), + "domain {domain} must be refused by the filter" + ); + } + assert!( + socket2::Socket::pair(socket2::Domain::UNIX, socket2::Type::STREAM, None).is_ok() + ); + assert_eq!( + socket2::Socket::pair( + socket2::Domain::from(libc::AF_TIPC), + socket2::Type::STREAM, + None + ) + .map(drop) + .map_err(|error| error.raw_os_error()), + Err(Some(libc::EPERM)) + ); + return; + } + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "sandbox::linux::seccomp::tests::behavioral_socket_families_are_allowlisted", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + .expect("run isolated socket family test"); + assert!(status.success(), "isolated socket family test failed"); + } + #[test] fn behavioral_block_mode_denies_inet_and_packet_sockets() { let filter = build_filter(false).unwrap(); diff --git a/docs/security/best-practices.mdx b/docs/security/best-practices.mdx index 2f5b346708..e70f38ecc4 100644 --- a/docs/security/best-practices.mdx +++ b/docs/security/best-practices.mdx @@ -220,7 +220,7 @@ OpenShell applies seccomp in two phases. A narrow supervisor-startup prelude run | Aspect | Detail | |---|---| | Startup prelude | After privileged bootstrap helpers complete, including network setup and provider-token SPIFFE child mount-namespace preparation, the supervisor sets `PR_SET_NO_NEW_PRIVS` and synchronizes a seccomp filter across all runtime threads that blocks `mount`, the new mount API syscalls, `pivot_root`, `umount2`, `bpf`, `perf_event_open`, `userfaultfd`, module-loading syscalls, and kexec. This closes the long-lived privileged remount and kernel-surface window while leaving required setup syscalls such as `setns` available. | -| Socket domains | The filter allows `AF_INET` and `AF_INET6` (for proxy communication) and blocks `AF_PACKET`, `AF_BLUETOOTH`, and `AF_VSOCK` with `EPERM`. `AF_NETLINK` is partially allowed: only `NETLINK_ROUTE` (protocol 0) is permitted so that `getifaddrs(3)` works; all other netlink protocols are blocked. Write operations via `NETLINK_ROUTE` still require `CAP_NET_ADMIN`, which the sandbox does not grant. | +| Socket domains | The filter allows only `AF_UNIX`, `AF_NETLINK`, and the brokered `AF_INET` and `AF_INET6` families. Every other family, including `AF_PACKET`, `AF_VSOCK`, `AF_RXRPC`, `AF_SMC`, `AF_KCM`, and `AF_ALG`, fails with `EPERM`, and `socketpair(2)` is limited to `AF_UNIX`. `AF_NETLINK` is partially allowed: only `NETLINK_ROUTE` (protocol 0) is permitted so that `getifaddrs(3)` works; all other netlink protocols are blocked. Write operations via `NETLINK_ROUTE` still require `CAP_NET_ADMIN`, which the sandbox does not grant. | | Runtime unconditional syscall blocks | `memfd_create`, `ptrace`, `bpf`, `process_vm_readv`, `process_vm_writev`, `pidfd_open`, `pidfd_getfd`, `pidfd_send_signal`, `io_uring_setup`, `mount`, `fsopen`, `fsconfig`, `fsmount`, `fspick`, `move_mount`, `open_tree`, `setns`, `umount2`, `pivot_root`, `userfaultfd`, `perf_event_open`. | | Conditional syscall blocks | `execveat` with `AT_EMPTY_PATH`, `unshare` and `clone` with `CLONE_NEWUSER`, and `seccomp(SECCOMP_SET_MODE_FILTER)` are denied with `EPERM`. | | What you can change | This is not a user-facing knob. OpenShell enforces it automatically. | From a5e46e149079759c7717a62d2e4124eefecf53b5 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 18:25:26 -0700 Subject: [PATCH 05/18] refactor(sandbox): remove legacy read-only mode After native local accept, only two broker paths still wrote into workload memory: getpeername on relayed connections and the per-message lengths of a first sendmmsg to the DNS relay. Legacy mode existed only to refuse those writes on kernels without WAIT_KILLABLE_RECV, and the getpeername refusal broke CPython TLS on RHEL 9. The broker now never writes workload memory: - getpeername is no longer mediated and reports the kernel peer on every kernel; relayed connections report the loopback relay address. - A first send to the DNS relay pins the broker's socket copy to the relay and continues the syscall, so the kernel performs the send and writes any per-message results. Remove the listener mode, the task-memory write path and probe, and the task_memory_write, cancellation, and task_memory_writes_disabled audit evidence fields. WAIT_KILLABLE_RECV is still used when available and is reported for diagnostics, but no mediation decision depends on it. Signed-off-by: Drew Newberry --- .../src/linux/seccomp_notify.rs | 151 ++---------- .../src/linux/task_memory.rs | 90 +------ .../src/boundary_protocol.rs | 27 +-- .../openshell-sandbox-backend/src/runtime.rs | 3 - .../openshell-sandbox/src/boundary_server.rs | 3 - crates/openshell-sandbox/src/main.rs | 74 +----- .../openshell-sandbox/src/network_broker.rs | 221 +----------------- docs/about/support-matrix.mdx | 28 +-- docs/kubernetes/openshift.mdx | 17 +- 9 files changed, 38 insertions(+), 576 deletions(-) diff --git a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs index fbd8b78f9c..b5985b1c35 100644 --- a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs +++ b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs @@ -170,34 +170,6 @@ impl NotificationProbeReport { } } -/// Cancellation posture a listener was installed with. -/// -/// `WAIT_KILLABLE_RECV` (Linux 5.19+) keeps the notified workload thread in a -/// kill-only wait so a non-fatal signal cannot resume the mediated syscall -/// after the broker has validated the notification. A plain listener has no -/// such guarantee, so it runs read-only: the broker must refuse every -/// task-memory *output* write to stay cancellation-safe. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ListenerMode { - /// Modern kernel: `WAIT_KILLABLE_RECV` active; full mediation including - /// task-memory output writes. - Killable, - /// Legacy kernel (< 5.19): plain listener; task-memory output writes are - /// disabled so a resumed syscall cannot race a broker write. - LegacyReadOnly, -} - -impl ListenerMode { - /// Stable identifier for qualification output and diagnostics. - #[must_use] - pub fn as_str(self) -> &'static str { - match self { - Self::Killable => "killable", - Self::LegacyReadOnly => "legacy_read_only", - } - } -} - /// Owned listener returned by `SECCOMP_FILTER_FLAG_NEW_LISTENER`. pub struct NotificationListener { fd: OwnedFd, @@ -205,17 +177,6 @@ pub struct NotificationListener { } impl NotificationListener { - /// Construct a listener from an already-owned notification descriptor in a - /// specific mode. Intended for tests that must exercise the legacy - /// read-only fail-closed paths without a `< 5.19` kernel. - #[must_use] - pub fn from_fd_with_mode(fd: OwnedFd, mode: ListenerMode) -> Self { - Self { - fd, - wait_killable_recv: matches!(mode, ListenerMode::Killable), - } - } - /// Raw listener descriptor for readiness integration and diagnostics. #[must_use] pub fn as_raw_fd(&self) -> RawFd { @@ -223,28 +184,16 @@ impl NotificationListener { } /// Whether the listener was installed with killable receive waits. + /// + /// The broker never writes workload memory, so no mediation decision + /// depends on this. It only changes signal behavior: with killable waits a + /// non-fatal signal cannot interrupt a syscall that is waiting for the + /// broker or supervisor. #[must_use] pub fn wait_killable_recv(&self) -> bool { self.wait_killable_recv } - /// The cancellation mode this listener was installed with. - #[must_use] - pub fn mode(&self) -> ListenerMode { - if self.wait_killable_recv { - ListenerMode::Killable - } else { - ListenerMode::LegacyReadOnly - } - } - - /// Whether broker task-memory output writes are disabled for this listener. - /// True exactly in `LegacyReadOnly` mode (no `WAIT_KILLABLE_RECV`). - #[must_use] - pub fn writes_disabled(&self) -> bool { - matches!(self.mode(), ListenerMode::LegacyReadOnly) - } - /// Receive the next kernel notification. pub fn receive(&self) -> io::Result { let mut raw = RawNotification::default(); @@ -272,34 +221,6 @@ impl NotificationListener { Ok(()) } - /// Write broker-produced output into the notified task's memory, closing - /// the validation-to-write race that a plain listener cannot. - /// - /// In `Killable` mode `WAIT_KILLABLE_RECV` keeps the notified workload - /// thread in a kill-only wait, so a non-fatal signal cannot resume the - /// mediated syscall between `validate_id` and this write. In - /// `LegacyReadOnly` mode (kernels < 5.19) there is no such guarantee: a - /// resumed syscall could repurpose the target buffer while the privileged - /// broker writes through the captured tid and pointer — via `/proc//mem` - /// even into pages the workload has since made read-only. There is no way to - /// close that window without the flag, so this fails closed (`EOPNOTSUPP`) - /// rather than racing. Callers must route every task-memory *output* write - /// through this method; input reads never write workload memory and are - /// unaffected. - pub fn write_task_output( - &self, - id: u64, - tid: u32, - address: u64, - data: &[u8], - ) -> io::Result<()> { - if self.writes_disabled() { - return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); - } - self.validate_id(id)?; - crate::linux::task_memory::write_exact(tid, address, data) - } - /// Return a successful scalar result to the notifying syscall. pub fn respond_value(&self, id: u64, value: i64) -> io::Result<()> { self.validate_id(id)?; @@ -407,15 +328,12 @@ pub fn install_listener(syscalls: &[i64]) -> io::Result { verify_notification_sizes()?; set_no_new_privileges()?; - // WAIT_KILLABLE_RECV (Linux 5.19+) keeps the *notified workload thread* in - // a kill-only wait while the broker services its syscall, so a non-fatal - // signal cannot resume the syscall and repurpose its buffers underneath a - // pending broker write. Kernels older than 5.19 (for example RHEL 9.x / - // 5.14 nodes) reject the flag with EINVAL. Rather than refusing to start - // there, fall back to a plain listener so the sandbox boots; the resulting - // listener records `wait_killable_recv = false`, and the broker then fails - // closed on every task-memory output write (see `write_task_output`) - // instead of racing them. Input mediation is unaffected. + // WAIT_KILLABLE_RECV (Linux 5.19+) keeps the notified workload thread in a + // kill-only wait while the broker services its syscall, so a non-fatal + // signal cannot interrupt a syscall waiting on a supervisor decision. + // Kernels older than 5.19 (for example RHEL 9.x / 5.14) reject the flag + // with EINVAL; fall back to a plain listener there. The broker never + // writes workload memory, so mediation is identical with either listener. match install_listener_with_flags(syscalls, true) { Ok(listener) => Ok(listener), Err(error) if error.raw_os_error() == Some(libc::EINVAL) => { @@ -443,7 +361,6 @@ pub fn install_workload_listener() -> io::Result { libc::SYS_sendto, libc::SYS_sendmsg, libc::SYS_sendmmsg, - libc::SYS_getpeername, libc::SYS_setsockopt, libc::SYS_kill, libc::SYS_tkill, @@ -1086,50 +1003,10 @@ mod tests { } #[test] - fn plain_listener_fails_closed_on_output_write() { - // A LegacyReadOnly listener (kernels < 5.19) must refuse every - // task-memory output write rather than race a resumed syscall. The - // guard short-circuits before touching the descriptor or workload - // memory, so a dup of stderr is a sufficient stand-in. - // SAFETY: dup takes one valid descriptor and returns a new descriptor - // or a negative error without modifying memory. - let duplicated = unsafe { libc::dup(libc::STDERR_FILENO) }; - assert!(duplicated >= 0, "duplicate stderr for validation test"); - // SAFETY: successful dup returned a new owned descriptor. - let listener = NotificationListener::from_fd_with_mode( - unsafe { OwnedFd::from_raw_fd(duplicated) }, - ListenerMode::LegacyReadOnly, - ); - assert!(listener.writes_disabled()); - assert_eq!(listener.mode(), ListenerMode::LegacyReadOnly); - let error = listener - .write_task_output(1, 0, 0, &[0_u8; 4]) - .expect_err("plain listener must reject output writes"); - assert_eq!(error.raw_os_error(), Some(libc::EOPNOTSUPP)); - } - - #[test] - fn real_plain_listener_disables_output_writes() { - // Deliberately install a plain NEW_LISTENER (no WAIT_KILLABLE_RECV) - // even on a modern CI kernel and prove the broker write path fails - // closed on the real listener object. + fn listener_records_killable_receive_waits() { set_no_new_privileges().expect("no_new_privs for listener install"); - let listener = install_listener_with_flags(&[libc::SYS_getppid], false) + let plain = install_listener_with_flags(&[libc::SYS_getppid], false) .expect("install plain listener"); - assert_eq!(listener.mode(), ListenerMode::LegacyReadOnly); - assert!(listener.writes_disabled()); - let error = listener - .write_task_output(1, 0, 0, &[0_u8; 4]) - .expect_err("plain listener must reject output writes"); - assert_eq!(error.raw_os_error(), Some(libc::EOPNOTSUPP)); - } - - #[test] - fn killable_listener_enables_output_writes() { - set_no_new_privileges().expect("no_new_privs for listener install"); - let listener = install_listener_with_flags(&[libc::SYS_getppid], true) - .expect("install killable listener"); - assert_eq!(listener.mode(), ListenerMode::Killable); - assert!(!listener.writes_disabled()); + assert!(!plain.wait_killable_recv()); } } diff --git a/crates/openshell-isolation-interface/src/linux/task_memory.rs b/crates/openshell-isolation-interface/src/linux/task_memory.rs index 1b37685b3b..ed34f234e8 100644 --- a/crates/openshell-isolation-interface/src/linux/task_memory.rs +++ b/crates/openshell-isolation-interface/src/linux/task_memory.rs @@ -64,53 +64,6 @@ pub fn read_exact(tid: u32, address: u64, destination: &mut [u8]) -> io::Result< } } -/// Write exactly all of `source` to `address` in `tid`. -/// -/// This is used only for syscall outputs such as `getpeername` and -/// `sendmmsg.msg_len`. Revalidate the notification, task generation, and -/// destination layout immediately before calling it. -pub fn write_exact(tid: u32, address: u64, source: &[u8]) -> io::Result<()> { - validate_request(tid, address, source.len())?; - let pid = libc::pid_t::try_from(tid) - .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "TID does not fit pid_t"))?; - let remote_address = usize::try_from(address).map_err(|_| { - io::Error::new( - io::ErrorKind::InvalidInput, - "remote address does not fit usize", - ) - })?; - let local = libc::iovec { - iov_base: source.as_ptr().cast_mut().cast(), - iov_len: source.len(), - }; - let remote = libc::iovec { - iov_base: remote_address as *mut libc::c_void, - iov_len: source.len(), - }; - - // SAFETY: the local iovec spans the caller-provided live buffer. The - // remote address is untrusted but bounded; the kernel validates that it is - // writable in the target process. - let copied = retry_eintr(|| unsafe { - libc::process_vm_writev( - pid, - std::ptr::addr_of!(local), - 1, - std::ptr::addr_of!(remote), - 1, - 0, - ) - }); - match copied { - Ok(copied) => require_exact(copied, source.len(), "task-memory write"), - Err(error) if syscall_profile_denied(&error) => { - write_exact_to_proc_mem(tid, address, source) - .map_err(|fallback| fallback_error("write", &error, fallback)) - } - Err(error) => Err(error), - } -} - fn syscall_profile_denied(error: &io::Error) -> bool { matches!( error.raw_os_error(), @@ -145,22 +98,14 @@ fn read_exact_from_proc_mem(tid: u32, address: u64, destination: &mut [u8]) -> i require_exact(copied, destination.len(), "proc task-memory read") } -fn write_exact_to_proc_mem(tid: u32, address: u64, source: &[u8]) -> io::Result<()> { - let file = std::fs::OpenOptions::new() - .write(true) - .open(format!("/proc/{tid}/mem"))?; - let copied = file.write_at(source, address)?; - require_exact(copied, source.len(), "proc task-memory write") -} - -/// Prove same-UID parent-to-child read and write access under the active Yama, -/// LSM, and outer seccomp posture. +/// Prove same-UID parent-to-child read access under the active Yama, LSM, and +/// outer seccomp posture. The broker only reads workload memory; it never +/// writes it. /// /// Call this only from a single-threaded probe process. The child executes /// raw, allocation-free syscalls between `fork` and `_exit`. pub fn probe_child_access() -> io::Result<()> { const INITIAL: u64 = 0x1122_3344_5566_7788; - const REPLACEMENT: u64 = 0xaabb_ccdd_eeff_0011; // SAFETY: mmap creates one private anonymous page owned by this process. let mapping = unsafe { libc::mmap( @@ -216,7 +161,6 @@ pub fn probe_child_access() -> io::Result<()> { if libc::prctl(libc::PR_SET_DUMPABLE, 1, 0, 0, 0) < 0 || write_eventfd(ready.as_raw_fd()).is_err() || read_eventfd(proceed.as_raw_fd()).is_err() - || mapping.cast::().read() != REPLACEMENT { libc::_exit(1); } @@ -237,11 +181,6 @@ pub fn probe_child_access() -> io::Result<()> { "cross-child memory read returned wrong data", )); } - write_exact( - u32::try_from(child).map_err(|_| io::Error::other("child PID does not fit u32"))?, - mapping_address, - &REPLACEMENT.to_ne_bytes(), - )?; write_eventfd(proceed.as_raw_fd())?; let mut status = 0; // SAFETY: child is a live direct child and status points to storage. @@ -354,9 +293,8 @@ mod tests { use super::*; #[test] - fn reads_and_writes_exact_same_process_memory() { + fn reads_exact_same_process_memory() { let source = 0x1122_3344_5566_7788_u64; - let mut destination = 0_u64; let mut bytes = [0_u8; size_of::()]; read_exact( @@ -366,21 +304,11 @@ mod tests { ) .expect("read source"); assert_eq!(u64::from_ne_bytes(bytes), source); - - let replacement = 0xaabb_ccdd_eeff_0011_u64; - write_exact( - std::process::id(), - std::ptr::addr_of_mut!(destination) as u64, - &replacement.to_ne_bytes(), - ) - .expect("write destination"); - assert_eq!(destination, replacement); } #[test] - fn proc_mem_fallback_reads_and_writes_exact_memory() { + fn proc_mem_fallback_reads_exact_memory() { let source = 0x0102_0304_0506_0708_u64; - let mut destination = 0_u64; let mut bytes = [0_u8; size_of::()]; read_exact_from_proc_mem( std::process::id(), @@ -389,14 +317,6 @@ mod tests { ) .expect("read through proc mem"); assert_eq!(u64::from_ne_bytes(bytes), source); - - write_exact_to_proc_mem( - std::process::id(), - std::ptr::addr_of_mut!(destination) as u64, - &source.to_ne_bytes(), - ) - .expect("write through proc mem"); - assert_eq!(destination, source); } #[test] diff --git a/crates/openshell-sandbox-backend/src/boundary_protocol.rs b/crates/openshell-sandbox-backend/src/boundary_protocol.rs index 5b46adbff1..369fe973d5 100644 --- a/crates/openshell-sandbox-backend/src/boundary_protocol.rs +++ b/crates/openshell-sandbox-backend/src/boundary_protocol.rs @@ -94,9 +94,6 @@ pub struct SeccompEvidence { pub retained_socket_operation: bool, pub proc_fd_identity: bool, pub task_memory_read: bool, - pub task_memory_write: bool, - pub cancellation: bool, - pub task_memory_writes_disabled: bool, } /// Mechanism-specific audit evidence for the native Linux sandbox adapter. @@ -147,8 +144,6 @@ impl NativeLinuxSandboxAuditEvidence { && self.seccomp.retained_socket_operation && self.seccomp.proc_fd_identity && self.seccomp.task_memory_read - && self.seccomp.task_memory_write - && (self.seccomp.cancellation || self.seccomp.task_memory_writes_disabled) && self.landlock_abi >= 3 && self.landlock_allow_deny && self.udp_dns_round_trip @@ -187,8 +182,7 @@ impl NativeLinuxSandboxAuditEvidence { request_attribution: EnforcedProperty::new( self.seccomp.id_validation && self.seccomp.proc_fd_identity - && self.seccomp.task_memory_read - && self.seccomp.task_memory_write, + && self.seccomp.task_memory_read, "seccomp-notify-procfs", ), privilege_floor: EnforcedProperty::new( @@ -1461,9 +1455,6 @@ mod tests { retained_socket_operation: true, proc_fd_identity: true, task_memory_read: true, - task_memory_write: true, - cancellation: true, - task_memory_writes_disabled: false, }, landlock_abi: 6, landlock_allow_deny: true, @@ -1513,22 +1504,6 @@ mod tests { assert!(serde_json::from_value::(value).is_err()); } - #[test] - fn audit_evidence_accepts_legacy_read_only_listener() { - let mut audit = complete_audit_evidence(); - audit.seccomp.cancellation = false; - audit.seccomp.task_memory_writes_disabled = true; - assert!(audit.validate().is_ok()); - } - - #[test] - fn audit_evidence_rejects_plain_listener_with_writes_enabled() { - let mut audit = complete_audit_evidence(); - audit.seccomp.cancellation = false; - audit.seccomp.task_memory_writes_disabled = false; - assert!(audit.validate().is_err()); - } - #[test] fn binary_identity_wire_rejects_ambiguous_or_invalid_shapes() { for encoded in [ diff --git a/crates/openshell-sandbox-backend/src/runtime.rs b/crates/openshell-sandbox-backend/src/runtime.rs index c1e36f1b4c..fea7c29d27 100644 --- a/crates/openshell-sandbox-backend/src/runtime.rs +++ b/crates/openshell-sandbox-backend/src/runtime.rs @@ -2906,9 +2906,6 @@ mod tests { retained_socket_operation: true, proc_fd_identity: true, task_memory_read: true, - task_memory_write: true, - cancellation: true, - task_memory_writes_disabled: false, }, landlock_abi: 3, landlock_allow_deny: true, diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index effd97dac2..4695d65b31 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -4541,9 +4541,6 @@ mod linux { retained_socket_operation: true, proc_fd_identity: true, task_memory_read: true, - task_memory_write: true, - cancellation: true, - task_memory_writes_disabled: false, }, landlock_abi: 6, landlock_allow_deny: true, diff --git a/crates/openshell-sandbox/src/main.rs b/crates/openshell-sandbox/src/main.rs index aaec0a0d11..f4b0ce6809 100644 --- a/crates/openshell-sandbox/src/main.rs +++ b/crates/openshell-sandbox/src/main.rs @@ -103,12 +103,9 @@ struct QualificationReport { tcp_dns_round_trip: bool, tcp_allow_round_trip: bool, tcp_deny_round_trip: bool, + /// Whether notified syscalls wait killably (Linux 5.19+). Informational: + /// the broker never writes workload memory, so mediation is identical. wait_killable_recv: bool, - /// Selected seccomp listener cancellation mode: `killable` (>= 5.19) or - /// `legacy_read_only` (< 5.19, broker output writes disabled). - seccomp_listener_mode: &'static str, - /// Whether the broker disables task-memory output writes (legacy mode). - task_memory_writes_disabled: bool, } #[cfg(target_os = "linux")] @@ -209,12 +206,6 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi tcp_allow_round_trip: true, tcp_deny_round_trip: true, wait_killable_recv: notification.wait_killable_recv, - seccomp_listener_mode: if notification.wait_killable_recv { - "killable" - } else { - "legacy_read_only" - }, - task_memory_writes_disabled: !notification.wait_killable_recv, }; let qualification = openshell_sandbox::RuntimeQualification { seccomp: openshell_sandbox_backend::boundary_protocol::SeccompEvidence { @@ -225,11 +216,6 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi retained_socket_operation: true, proc_fd_identity: true, task_memory_read: task_memory_copy, - task_memory_write: task_memory_copy, - cancellation: notification.wait_killable_recv, - // Legacy plain listener (< 5.19) disables broker output writes; - // satisfies the `cancellation || writes_disabled` launch invariant. - task_memory_writes_disabled: !notification.wait_killable_recv, }, landlock_abi, landlock_allow_deny: true, @@ -469,7 +455,6 @@ fn probe_socket_virtualization() -> Result<()> { openshell_isolation_interface::linux::seccomp_notify::install_listener(&[ libc::SYS_socket, libc::SYS_connect, - libc::SYS_getpeername, libc::SYS_sendto, ]); let Ok(listener) = listener else { @@ -524,7 +509,6 @@ fn probe_socket_virtualization() -> Result<()> { let mut observed_connect = false; let mut observed_dns_tcp_connect = false; let mut observed_denied_connect = false; - let mut observed_peer = false; let mut observed_dns_send = false; while !(observed_tcp_sockets == 3 @@ -532,7 +516,6 @@ fn probe_socket_virtualization() -> Result<()> { && observed_connect && observed_dns_tcp_connect && observed_denied_connect - && observed_peer && observed_dns_send) { let notification = listener @@ -658,27 +641,6 @@ fn probe_socket_virtualization() -> Result<()> { .respond_value(notification.id, 0) .into_diagnostic()?; } - libc::SYS_getpeername => { - let fd = i32::try_from(notification.args[0]) - .map_err(|_| miette::miette!("peer FD does not fit i32"))?; - let entry = registry.resolve(notification.tid, fd).into_diagnostic()?; - let SocketState::Connected { original_peer } = entry.state() else { - listener - .respond_errno(notification.id, libc::ENOTCONN) - .into_diagnostic()?; - return Err(miette::miette!("peer query preceded mediated connect")); - }; - write_probe_sockaddr( - notification.tid, - notification.args[1], - notification.args[2], - *original_peer, - )?; - listener - .respond_value(notification.id, 0) - .into_diagnostic()?; - observed_peer = true; - } libc::SYS_sendto => { let fd = i32::try_from(notification.args[0]) .map_err(|_| miette::miette!("sendto FD does not fit i32"))?; @@ -952,38 +914,6 @@ fn decode_probe_sockaddr(bytes: &[u8]) -> Result { ))) } -#[cfg(target_os = "linux")] -fn write_probe_sockaddr( - tid: u32, - address: u64, - length_address: u64, - peer: std::net::SocketAddr, -) -> Result<()> { - use std::mem::size_of; - - let (sockaddr, sockaddr_length) = encode_probe_sockaddr(peer)?; - let mut requested_length = [0_u8; size_of::()]; - openshell_isolation_interface::linux::task_memory::read_exact( - tid, - length_address, - &mut requested_length, - ) - .into_diagnostic()?; - let requested_length = libc::socklen_t::from_ne_bytes(requested_length); - if requested_length < sockaddr_length { - return Err(miette::miette!("peer sockaddr buffer is too small")); - } - openshell_isolation_interface::linux::task_memory::write_exact(tid, address, &sockaddr) - .into_diagnostic()?; - openshell_isolation_interface::linux::task_memory::write_exact( - tid, - length_address, - &sockaddr_length.to_ne_bytes(), - ) - .into_diagnostic()?; - Ok(()) -} - #[cfg(target_os = "linux")] #[allow(unsafe_code)] fn run_capability_socket_child(args: &[String]) -> Result<()> { diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 5be49d4c21..f1671c6e9b 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -547,9 +547,6 @@ fn dispatch_notification( ) { return classify_send(®istry, &listener, notification, &queues.dns_relay); } - if syscall == libc::SYS_getpeername { - return get_peer_name(®istry, &listener, notification); - } if syscall == libc::SYS_setsockopt { let level = i32::try_from(notification.args[1]) .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; @@ -1115,9 +1112,6 @@ fn classify_send( libc::SYS_sendmsg => vec![read_sendmsg_message( notification.tid, notification.args[1], - i32::try_from(notification.args[2]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, - None, )?], libc::SYS_sendmmsg => read_sendmmsg_messages(notification)?, _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), @@ -1177,29 +1171,14 @@ fn classify_send( lock(&dns_relay.udp_admissions).remove(&peer); return Err(error); } - for message in &messages { - send_dns_message(source_fd, message)?; - if let Some(length_address) = message.result_length_address { - let length = u32::try_from(message.data.len()) - .map_err(|_| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - listener.write_task_output( - notification.id, - notification.tid, - length_address, - &length.to_ne_bytes(), - )?; - } - } entry.set_state(SocketState::DnsUdp { relay: dns_relay.address, }); entry.release_preconnect(); - let result = if syscall == libc::SYS_sendmmsg { - i64::try_from(messages.len()).unwrap_or(i64::MAX) - } else { - i64::try_from(messages[0].data.len()).unwrap_or(i64::MAX) - }; - listener.respond_value(notification.id, result) + // The socket is now pinned to the relay and bound to loopback. The + // kernel performs the send and writes any per-message results, so + // the broker never writes workload memory. + listener.respond_continue(notification.id) } Ok(_) => Err(io::Error::from_raw_os_error(libc::EDESTADDRREQ)), // Non-INET sockets and natively accepted sockets were never @@ -1226,20 +1205,10 @@ fn send_flags(syscall: i64, args: [u64; 6]) -> i32 { } struct SendMessage { - data: Vec, destination: Option, - flags: i32, - result_length_address: Option, } fn read_sendto_message(notification: Notification) -> io::Result { - let length = usize::try_from(notification.args[2]) - .map_err(|_| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - if u16::try_from(length).is_err() { - return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); - } - let mut data = vec![0_u8; length]; - task_memory::read_exact(notification.tid, notification.args[1], &mut data)?; let destination = if notification.args[4] == 0 { None } else { @@ -1249,21 +1218,10 @@ fn read_sendto_message(notification: Notification) -> io::Result { notification.args[5], )?) }; - Ok(SendMessage { - data, - destination, - flags: i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, - result_length_address: None, - }) + Ok(SendMessage { destination }) } -fn read_sendmsg_message( - tid: u32, - address: u64, - flags: i32, - result_length_address: Option, -) -> io::Result { +fn read_sendmsg_message(tid: u32, address: u64) -> io::Result { let header = read_task_value::(tid, address)?; if header.msg_controllen != 0 { return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); @@ -1277,39 +1235,7 @@ fn read_sendmsg_message( u64::from(header.msg_namelen), )?) }; - #[cfg(target_env = "musl")] - let iov_count = usize::try_from(header.msg_iovlen) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; - #[cfg(not(target_env = "musl"))] - let iov_count = header.msg_iovlen; - if iov_count > 32 { - return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); - } - let mut data = Vec::new(); - for index in 0..iov_count { - let offset = index - .checked_mul(size_of::()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?; - let iov = read_task_value::( - tid, - (header.msg_iov as u64) - .checked_add(u64::try_from(offset).unwrap_or(u64::MAX)) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?, - )?; - let start = data.len(); - let end = start - .checked_add(iov.iov_len) - .filter(|length| u16::try_from(*length).is_ok()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - data.resize(end, 0); - task_memory::read_exact(tid, iov.iov_base as u64, &mut data[start..end])?; - } - Ok(SendMessage { - data, - destination, - flags, - result_length_address, - }) + Ok(SendMessage { destination }) } fn read_sendmmsg_messages(notification: Notification) -> io::Result> { @@ -1318,8 +1244,6 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result 32 { return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); } - let flags = i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; (0..count) .map(|index| { let offset = index @@ -1328,18 +1252,7 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result(tid: u32, address: u64) -> io::Result { Ok(unsafe { std::ptr::read_unaligned(bytes.as_ptr().cast::()) }) } -fn send_dns_message(fd: RawFd, message: &SendMessage) -> io::Result<()> { - // SAFETY: `fd` is the retained exact UDP socket and the buffer remains - // valid for the duration of the syscall. - let sent = unsafe { - libc::send( - fd, - message.data.as_ptr().cast(), - message.data.len(), - message.flags, - ) - }; - if sent < 0 { - return Err(io::Error::last_os_error()); - } - if usize::try_from(sent).ok() == Some(message.data.len()) { - Ok(()) - } else { - Err(io::Error::from_raw_os_error(libc::EIO)) - } -} - -fn get_peer_name( - registry: &Mutex, - listener: &NotificationListener, - notification: Notification, -) -> io::Result<()> { - let fd = raw_fd(notification.args[0])?; - let registry = lock(registry); - let Ok(entry) = registry.resolve(notification.tid, fd) else { - return listener.respond_continue(notification.id); - }; - // Only relayed sockets need a synthesized original peer. Every other - // descriptor reports its true kernel peer, including on legacy listeners - // where the broker cannot write into workload memory. Continuing on a - // substituted descriptor only discloses that descriptor's own peer. - let SocketState::Connected { original_peer } = entry.state() else { - return listener.respond_continue(notification.id); - }; - let peer = *original_peer; - write_socket_addr( - listener, - notification.id, - notification.tid, - notification.args[1], - notification.args[2], - peer, - )?; - listener.respond_value(notification.id, 0) -} - fn connect_exact(fd: RawFd, address: SocketAddr) -> io::Result<()> { // Never let a blocking connect pin the single notification dispatcher. // O_NONBLOCK is an OFD flag, so restore the workload's original setting @@ -1548,49 +1411,6 @@ fn decode_sockaddr(storage: libc::sockaddr_storage, length: usize) -> io::Result } } -fn write_socket_addr( - listener: &NotificationListener, - notification_id: u64, - tid: u32, - address: u64, - length_address: u64, - value: SocketAddr, -) -> io::Result<()> { - // A LegacyReadOnly listener (kernels < 5.19) cannot safely write into - // workload memory: without WAIT_KILLABLE_RECV the notified getpeername - // could resume and repurpose these buffers between validation and the - // broker write. Fail closed before reading or writing anything, so this - // address-writing path is inert in legacy mode. - if listener.writes_disabled() { - return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); - } - let mut supplied_length = [0_u8; size_of::()]; - task_memory::read_exact(tid, length_address, &mut supplied_length)?; - let supplied_length = libc::socklen_t::from_ne_bytes(supplied_length); - let (bytes, actual_length) = sockaddr_bytes(value)?; - let copied = usize::try_from(supplied_length) - .unwrap_or(0) - .min(bytes.len()); - if copied != 0 { - listener.write_task_output(notification_id, tid, address, &bytes[..copied])?; - } - listener.write_task_output( - notification_id, - tid, - length_address, - &actual_length.to_ne_bytes(), - ) -} - -fn sockaddr_bytes(address: SocketAddr) -> io::Result<(Vec, libc::socklen_t)> { - with_sockaddr(address, |native, length| { - let length_usize = usize::try_from(length).map_err(io::Error::other)?; - // SAFETY: with_sockaddr lends fully initialized storage for this call. - let bytes = unsafe { std::slice::from_raw_parts(native.cast::(), length_usize) }; - Ok((bytes.to_vec(), length)) - }) -} - fn with_sockaddr( address: SocketAddr, operation: impl FnOnce(*const libc::sockaddr, libc::socklen_t) -> io::Result, @@ -1648,7 +1468,6 @@ fn error_to_errno(error: &io::Error) -> i32 { #[cfg(test)] mod tests { use super::*; - use openshell_isolation_interface::linux::seccomp_notify::ListenerMode; use openshell_isolation_interface::linux::socket_confinement; #[test] @@ -1829,25 +1648,6 @@ mod tests { ))); } - #[test] - fn legacy_listener_rejects_socket_addr_write() { - // Relayed getpeername routes through write_socket_addr; on a - // LegacyReadOnly listener the path must fail closed (EOPNOTSUPP) - // before any task-memory access. - // SAFETY: dup returns a new descriptor or a negative error. - let dup = unsafe { libc::dup(libc::STDERR_FILENO) }; - assert!(dup >= 0, "dup stderr"); - let listener = NotificationListener::from_fd_with_mode( - // SAFETY: successful dup returned a new owned descriptor. - unsafe { OwnedFd::from_raw_fd(dup) }, - ListenerMode::LegacyReadOnly, - ); - let peer: SocketAddr = "127.0.0.1:8080".parse().unwrap(); - let error = write_socket_addr(&listener, 1, 0, 0, 0, peer) - .expect_err("legacy listener must reject socket-address writes"); - assert_eq!(error.raw_os_error(), Some(libc::EOPNOTSUPP)); - } - #[test] fn relay_rejects_descriptor_replaced_after_policy_decision() { let metadata = SocketMetadata { @@ -2198,9 +1998,8 @@ mod tests { ready_tx .send(listener.local_addr()?) .map_err(|_| io::Error::other("test client disappeared"))?; - // std passes a peer-address buffer, which the broker - // could not fill on a legacy listener. Native accept - // reports it directly from the kernel. + // std passes a peer-address buffer; native accept + // fills it directly from the kernel. let (stream, accepted_peer) = listener.accept()?; let peer = stream.peer_addr()?; let device = socket_confinement::bound_device(&stream)?; diff --git a/docs/about/support-matrix.mdx b/docs/about/support-matrix.mdx index 4c87744e49..cfc2fc212c 100644 --- a/docs/about/support-matrix.mdx +++ b/docs/about/support-matrix.mdx @@ -170,9 +170,9 @@ when it runs inside a container or microVM: | -------------------------------------------------------------- | ----------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | [Landlock LSM](https://docs.kernel.org/security/landlock.html) | Required | ABI 3 or newer, introduced in Linux 6.2, with Landlock enabled. The mandatory baseline protects private channel and bootstrap files, including against truncation. A filesystem policy's `best_effort` setting never disables this baseline. | | seccomp | Required | Nested user-notification filters and atomic `SECCOMP_IOCTL_NOTIF_ADDFD` with `SECCOMP_ADDFD_FLAG_SEND`, usable under the runtime's existing seccomp profile without added capabilities. The sandbox actively probes these operations before admitting the workload. | -| Task-memory access | Required | The non-dumpable broker must be able to read and write a same-UID, dumpable workload child's memory through `process_vm_readv` / `process_vm_writev` or `/proc//mem`. The sandbox actively probes the production parent-to-child topology before admitting the workload. | +| Task-memory access | Required | The non-dumpable broker must be able to read a same-UID, dumpable workload child's memory through `process_vm_readv` or `/proc//mem`. The broker never writes workload memory. The sandbox actively probes the production parent-to-child topology before admitting the workload. | | Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE` before handing it to the workload, without added capabilities (Linux 5.14+). Accepted sockets inherit the binding, so `accept` runs natively and only admits clients in the sandbox's own network namespace. Such a client that is not a workload process may present a non-loopback source address. No broker limit applies to accepted connections; each process's open-file limit (`RLIMIT_NOFILE`), the runtime PID limit, and the sandbox memory limit bound them. The sandbox actively probes that the binding installs, cannot be cleared, and is inherited before admitting the workload. | -| seccomp `WAIT_KILLABLE_RECV` | Recommended (Linux 5.19+) | Keeps a notified workload thread in a kill-only wait so the broker can safely write mediated results into workload memory. Without it (kernels < 5.19, for example RHEL 9.x / RHCOS 5.14) the sandbox still starts, in a reduced **legacy read-only** mode described below. | +| seccomp `WAIT_KILLABLE_RECV` | Optional (Linux 5.19+) | Used when available so a non-fatal signal cannot interrupt a syscall that is waiting for a mediation decision. Mediation behaves the same with or without it, including on RHEL 9.x / RHCOS 5.14. | A kernel version alone does not establish support. A disabled Landlock LSM or a runtime profile that blocks the required seccomp operations causes launch to @@ -180,31 +180,9 @@ fail closed. An upstream Linux 6.2 or newer kernel provides the required Landlock ABI; distribution backports must pass the same active qualification. The broker remains non-dumpable during qualification. A runtime may satisfy task-memory access through `/proc//mem` even when its kernel omits the -`process_vm_readv` and `process_vm_writev` system calls; OpenShell qualifies the +`process_vm_readv` system call; OpenShell qualifies the same parent-to-executed-child access shape used by mediated workloads. -### Legacy read-only mode (kernels before Linux 5.19) - -`SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` was added in Linux 5.19. On older -kernels — notably RHEL 9.x and RHCOS, which ship a 5.14 kernel — the sandbox -cannot install a kill-only listener, so it falls back to a plain listener and -runs in a **legacy read-only** cancellation mode. The sandbox starts and -enforces the full isolation boundary (Landlock, the outer NetworkPolicy fence, -DNS and TCP authorization); the only difference is that the broker refuses the -mediated operations that write results back into workload memory, failing them -closed with `EOPNOTSUPP`: - -- `getpeername` on connections relayed through the supervisor, which would - report the original upstream address; -- `sendmmsg` paths that write per-message lengths back to the caller. - -Socket creation, `connect`, `bind`, `listen`, `accept`, `sendto`, `sendmsg`, -and `getpeername` on local connections are unaffected. Local server workloads, -including those that read the peer address on accept, run unchanged. The -selected mode is reported in the sandbox -qualification output as `seccomp_listener_mode` (`killable` or -`legacy_read_only`). - On macOS, these kernel modules run inside the Docker Desktop Linux VM, not on the host kernel. ## Agent Workloads diff --git a/docs/kubernetes/openshift.mdx b/docs/kubernetes/openshift.mdx index 1c048e2628..191ad3521a 100644 --- a/docs/kubernetes/openshift.mdx +++ b/docs/kubernetes/openshift.mdx @@ -19,23 +19,12 @@ process to install a nested seccomp user-notification filter and use Landlock. OpenShell fails sandbox startup when either capability-free runtime probe fails. -## Node kernel and legacy read-only mode +## Node kernel OpenShell requires OpenShift 4.19 or later. The RHCOS kernels in OpenShift 4.16 through 4.18 (RHEL 9.4, 5.14.0-427) are built without Landlock, so sandbox -startup fails its Landlock probe on those releases. - -OpenShift nodes run RHCOS, which currently ships a RHEL 9.x kernel (5.14). That -kernel predates `SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` (Linux 5.19), so the -sandbox starts in a reduced **legacy read-only** cancellation mode. Isolation is -unchanged, but the broker fails closed with `EOPNOTSUPP` on the mediated -operations that write results back into workload memory: `getpeername` on -connections relayed through the supervisor, and `sendmmsg` per-message length -write-backs. Local servers, including `accept` with a peer address, run -unchanged. See the -[support matrix](/about/support-matrix#legacy-read-only-mode-kernels-before-linux-519) -for the full behavior; the selected mode is reported as `seccomp_listener_mode` -in the sandbox qualification output. +startup fails its Landlock probe on those releases. On supported releases the +sandbox behaves the same as on newer upstream kernels. ## Prerequisites From 4d2a12ef5b632657532d74928d012519717ce0cf Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 18:39:35 -0700 Subject: [PATCH 06/18] refactor(sandbox): drop WAIT_KILLABLE_RECV and make handlers restart-safe The broker no longer writes workload memory, so killable notification waits only reduced how often a signal restarts a notified syscall. RHEL 9 kernels never had them, so the handlers must tolerate restarts anyway. Install the listener without the flag on every kernel instead of special-casing newer ones. Handlers now check that the notification is still live immediately before each side effect (bind, connect of the retained socket, listen, and the first DNS send), and answer a restarted operation the broker already completed as the kernel would: a repeated TCP connect returns EISCONN, repeating the same UDP association succeeds, a repeat of a completed bind succeeds, and a failed relay reports its errno. Signed-off-by: Drew Newberry --- .../src/linux/seccomp_notify.rs | 72 ++-------- crates/openshell-sandbox/src/main.rs | 4 - .../openshell-sandbox/src/network_broker.rs | 131 +++++++++++++++++- docs/about/support-matrix.mdx | 1 - 4 files changed, 144 insertions(+), 64 deletions(-) diff --git a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs index b5985b1c35..6cf48abd4d 100644 --- a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs +++ b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs @@ -20,7 +20,6 @@ use std::time::Duration; const SECCOMP_SET_MODE_FILTER: libc::c_uint = 1; const SECCOMP_GET_NOTIF_SIZES: libc::c_uint = 3; const SECCOMP_FILTER_FLAG_NEW_LISTENER: libc::c_ulong = 1 << 3; -const SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV: libc::c_ulong = 1 << 5; const SECCOMP_RET_KILL_PROCESS: u32 = 0x8000_0000; const SECCOMP_RET_USER_NOTIF: u32 = 0x7fc0_0000; @@ -141,8 +140,6 @@ pub struct Notification { /// kernel, outer seccomp profile, and LSM posture. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct NotificationProbeReport { - /// Whether `SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` was accepted. - pub wait_killable_recv: bool, features: NotificationProbeFeatures, } @@ -171,9 +168,13 @@ impl NotificationProbeReport { } /// Owned listener returned by `SECCOMP_FILTER_FLAG_NEW_LISTENER`. +/// +/// The listener is installed without `WAIT_KILLABLE_RECV`, so mediation is the +/// same on every kernel. A signal can interrupt a notified syscall and the +/// kernel then restarts it; broker handlers check the notification is still +/// live before acting and answer a repeated operation as the kernel would. pub struct NotificationListener { fd: OwnedFd, - wait_killable_recv: bool, } impl NotificationListener { @@ -183,17 +184,6 @@ impl NotificationListener { self.fd.as_raw_fd() } - /// Whether the listener was installed with killable receive waits. - /// - /// The broker never writes workload memory, so no mediation decision - /// depends on this. It only changes signal behavior: with killable waits a - /// non-fatal signal cannot interrupt a syscall that is waiting for the - /// broker or supervisor. - #[must_use] - pub fn wait_killable_recv(&self) -> bool { - self.wait_killable_recv - } - /// Receive the next kernel notification. pub fn receive(&self) -> io::Result { let mut raw = RawNotification::default(); @@ -328,19 +318,7 @@ pub fn install_listener(syscalls: &[i64]) -> io::Result { verify_notification_sizes()?; set_no_new_privileges()?; - // WAIT_KILLABLE_RECV (Linux 5.19+) keeps the notified workload thread in a - // kill-only wait while the broker services its syscall, so a non-fatal - // signal cannot interrupt a syscall waiting on a supervisor decision. - // Kernels older than 5.19 (for example RHEL 9.x / 5.14) reject the flag - // with EINVAL; fall back to a plain listener there. The broker never - // writes workload memory, so mediation is identical with either listener. - match install_listener_with_flags(syscalls, true) { - Ok(listener) => Ok(listener), - Err(error) if error.raw_os_error() == Some(libc::EINVAL) => { - install_listener_with_flags(syscalls, false) - } - Err(error) => Err(error), - } + install_notification_filter(syscalls) } /// Install the capability-free workload networking listener on the calling @@ -379,30 +357,28 @@ pub fn install_workload_listener() -> io::Result { /// one dedicated thread and moved to an unfiltered broker thread through an /// in-process channel. pub fn probe_notification_api() -> io::Result { - let wait_killable_recv = probe_scalar_round_trip()?; + probe_scalar_round_trip()?; probe_addfd_send()?; probe_connected_sendto_fast_path()?; Ok(NotificationProbeReport { - wait_killable_recv, features: NotificationProbeFeatures(1 | 2 | 4), }) } -fn probe_scalar_round_trip() -> io::Result { +fn probe_scalar_round_trip() -> io::Result<()> { const PROBE_VALUE: libc::c_long = 0x5a17; let (sender, receiver) = mpsc::sync_channel(1); let launcher = thread::spawn(move || -> io::Result { let listener = install_listener(&[libc::SYS_getppid])?; - let wait_killable = listener.wait_killable_recv(); sender - .send((listener, wait_killable)) + .send(listener) .map_err(|_| io::Error::other("notification broker disappeared"))?; // SAFETY: getppid has no pointer arguments. The installed filter causes // the kernel to block here until the broker validates and responds. Ok(unsafe { libc::syscall(libc::SYS_getppid) }) }); - let (listener, wait_killable) = receiver + let listener = receiver .recv() .map_err(|_| io::Error::other("notification launcher disappeared"))?; let notification = match receive_probe_notification(&listener) { @@ -422,7 +398,7 @@ fn probe_scalar_round_trip() -> io::Result { if observed != PROBE_VALUE { return Err(io::Error::other("seccomp response value was not delivered")); } - Ok(wait_killable) + Ok(()) } fn probe_addfd_send() -> io::Result<()> { @@ -676,10 +652,7 @@ fn receive_probe_notification(listener: &NotificationListener) -> io::Result io::Result { +fn install_notification_filter(syscalls: &[i64]) -> io::Result { let mut program = build_filter(syscalls)?; let length = u16::try_from(program.len()) .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "seccomp filter is too large"))?; @@ -687,12 +660,7 @@ fn install_listener_with_flags( len: length, filter: program.as_mut_ptr(), }; - let flags = SECCOMP_FILTER_FLAG_NEW_LISTENER - | if wait_killable_recv { - SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV - } else { - 0 - }; + let flags = SECCOMP_FILTER_FLAG_NEW_LISTENER; // SAFETY: `fprog` points to a live classic-BPF program for the duration of // the syscall. The returned nonnegative value is a newly owned FD. let result = unsafe { @@ -710,10 +678,7 @@ fn install_listener_with_flags( .map_err(|_| io::Error::other("seccomp listener FD does not fit RawFd"))?; // SAFETY: successful NEW_LISTENER returns one newly owned descriptor. let fd = unsafe { OwnedFd::from_raw_fd(fd) }; - Ok(NotificationListener { - fd, - wait_killable_recv, - }) + Ok(NotificationListener { fd }) } fn build_filter(syscalls: &[i64]) -> io::Result> { @@ -994,19 +959,10 @@ mod tests { let listener = NotificationListener { // SAFETY: successful dup returned a new owned descriptor. fd: unsafe { OwnedFd::from_raw_fd(duplicated) }, - wait_killable_recv: false, }; let error = listener .respond_errno(1, 0) .expect_err("zero errno must fail"); assert_eq!(error.kind(), io::ErrorKind::InvalidInput); } - - #[test] - fn listener_records_killable_receive_waits() { - set_no_new_privileges().expect("no_new_privs for listener install"); - let plain = install_listener_with_flags(&[libc::SYS_getppid], false) - .expect("install plain listener"); - assert!(!plain.wait_killable_recv()); - } } diff --git a/crates/openshell-sandbox/src/main.rs b/crates/openshell-sandbox/src/main.rs index f4b0ce6809..78f6fa514b 100644 --- a/crates/openshell-sandbox/src/main.rs +++ b/crates/openshell-sandbox/src/main.rs @@ -103,9 +103,6 @@ struct QualificationReport { tcp_dns_round_trip: bool, tcp_allow_round_trip: bool, tcp_deny_round_trip: bool, - /// Whether notified syscalls wait killably (Linux 5.19+). Informational: - /// the broker never writes workload memory, so mediation is identical. - wait_killable_recv: bool, } #[cfg(target_os = "linux")] @@ -205,7 +202,6 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, - wait_killable_recv: notification.wait_killable_recv, }; let qualification = openshell_sandbox::RuntimeQualification { seccomp: openshell_sandbox_backend::boundary_protocol::SeccompEvidence { diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index f1671c6e9b..93ed821966 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -761,15 +761,21 @@ fn connect_socket( let destination = read_socket_addr(notification.tid, notification.args[1], notification.args[2])?; reject_protected_control_destination(destination, protected_control_port)?; - let (kind, socket_identity, nonblocking) = { + let (kind, socket_identity, nonblocking, repeated) = { let registry = lock(®istry); let entry = registry.resolve(notification.tid, fd)?; ( entry.metadata().kind, entry.identity(), entry.metadata().nonblocking, + repeated_connect_outcome(entry.state(), entry.metadata().kind, destination), ) }; + match repeated { + Some(0) => return listener.respond_value(notification.id, 0), + Some(errno) => return Err(io::Error::from_raw_os_error(errno)), + None => {} + } if kind == InetKind::DnsUdp && destination.port() == 0 { let mut registry = lock(®istry); let entry = registry.resolve_mut(notification.tid, fd)?; @@ -786,6 +792,7 @@ fn connect_socket( if entry.metadata().family != destination_family { return Err(io::Error::from_raw_os_error(libc::EAFNOSUPPORT)); } + listener.validate_id(notification.id)?; // glibc and uv use UDP connect(..., port 0), getsockname(), and an // AF_UNSPEC disconnect to rank resolved addresses. Bind only to the // matching loopback family and report success; never connect the @@ -808,6 +815,7 @@ fn connect_socket( return Err(io::Error::from_raw_os_error(libc::EISCONN)); } let source_fd = entry.retained_preconnect()?.as_raw_fd(); + listener.validate_id(notification.id)?; let peer = ensure_dns_source_bound(source_fd, entry.metadata().family)?; let admissions = match kind { InetKind::Tcp => &dns_relay.tcp_admissions, @@ -832,6 +840,7 @@ fn connect_socket( { let mut registry = lock(®istry); let entry = registry.resolve_mut(notification.tid, fd)?; + listener.validate_id(notification.id)?; connect_exact(entry.retained_preconnect()?.as_raw_fd(), destination)?; entry.set_state(SocketState::Local { peer: destination }); entry.release_preconnect(); @@ -916,6 +925,29 @@ fn connect_socket( Ok(()) } +/// Result for a `connect` on a socket the broker already connected, as the +/// kernel would report it: `Some(0)` for success, `Some(errno)` for an error, +/// `None` when the socket is not yet connected. +/// +/// A notified syscall interrupted by a signal is restarted after the broker +/// may already have completed it, so a repeat must not depend on the broker's +/// released pre-connect descriptor. +fn repeated_connect_outcome( + state: &SocketState, + kind: InetKind, + destination: SocketAddr, +) -> Option { + match state { + SocketState::Created | SocketState::Bound { .. } | SocketState::Listening { .. } => None, + SocketState::Failed { errno } => Some(*errno), + // UDP connect replaces the association; repeating the same one succeeds. + SocketState::DnsUdp { relay } if *relay == destination => Some(0), + SocketState::Local { peer } if kind == InetKind::DnsUdp && *peer == destination => Some(0), + _ if kind == InetKind::Tcp => Some(libc::EISCONN), + _ => None, + } +} + const fn tcp_denial_errno(reason: TcpOpenDenial) -> i32 { match reason { TcpOpenDenial::PolicyDenied @@ -1025,6 +1057,12 @@ fn bind_socket( let bind_result = { let mut registry = lock(registry); let entry = registry.resolve_mut(notification.tid, fd)?; + // A native bind is never restarted, but a notified one can be after a + // signal. Report a repeat of the bind the broker completed as success. + if entry.state() == &(SocketState::Bound { local }) { + return listener.respond_value(notification.id, 0); + } + listener.validate_id(notification.id)?; bind_exact(entry.retained_preconnect()?.as_raw_fd(), local) }; if bind_result @@ -1070,6 +1108,8 @@ fn listen_socket( let Ok(entry) = registry.resolve_mut(notification.tid, fd) else { return listener.respond_continue(notification.id); }; + // listen(2) may be repeated natively, so a restart needs no special case. + listener.validate_id(notification.id)?; // SAFETY: retained descriptor is the exact registered socket OFD. if unsafe { libc::listen(entry.retained_preconnect()?.as_raw_fd(), backlog) } < 0 { return Err(io::Error::last_os_error()); @@ -1165,6 +1205,7 @@ fn classify_send( { let entry = registry.resolve_mut(notification.tid, fd)?; let source_fd = entry.retained_preconnect()?.as_raw_fd(); + listener.validate_id(notification.id)?; let peer = ensure_dns_source_bound(source_fd, entry.metadata().family)?; register_dns_socket(&dns_relay.udp_admissions, peer, entry.identity())?; if let Err(error) = connect_exact(source_fd, dns_relay.address) { @@ -2261,6 +2302,94 @@ mod tests { ); } + #[test] + fn repeated_connects_report_what_the_kernel_would() { + let relay: SocketAddr = "127.0.0.53:53".parse().unwrap(); + let peer: SocketAddr = "127.0.0.1:8080".parse().unwrap(); + let other: SocketAddr = "127.0.0.1:9090".parse().unwrap(); + for (state, kind, destination, expected) in [ + (SocketState::Created, InetKind::Tcp, peer, None), + ( + SocketState::Bound { local: peer }, + InetKind::Tcp, + peer, + None, + ), + ( + SocketState::Local { peer }, + InetKind::Tcp, + peer, + Some(libc::EISCONN), + ), + ( + SocketState::Connected { + original_peer: "203.0.113.7:443".parse().unwrap(), + }, + InetKind::Tcp, + peer, + Some(libc::EISCONN), + ), + ( + SocketState::DnsTcp { relay }, + InetKind::Tcp, + relay, + Some(libc::EISCONN), + ), + ( + SocketState::DnsUdp { relay }, + InetKind::DnsUdp, + relay, + Some(0), + ), + (SocketState::DnsUdp { relay }, InetKind::DnsUdp, other, None), + (SocketState::Local { peer }, InetKind::DnsUdp, peer, Some(0)), + ( + SocketState::Failed { + errno: libc::ECONNRESET, + }, + InetKind::Tcp, + peer, + Some(libc::ECONNRESET), + ), + ] { + assert_eq!( + repeated_connect_outcome(&state, kind, destination), + expected, + "{state:?} {kind:?} {destination}" + ); + } + } + + #[test] + fn restarted_bind_and_connect_are_answered_consistently() { + // Without killable notification waits a signal can restart a + // syscall the broker already completed. A repeat must not fail on + // the broker's released pre-connect descriptor. + let service = TcpListener::bind("127.0.0.1:0").unwrap(); + let address = service.local_addr().unwrap(); + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let (bind_again, connect_again) = launcher + .execute(move || -> io::Result<(io::Result<()>, Option)> { + let socket = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None)?; + let local: SocketAddr = "127.0.0.1:0".parse().unwrap(); + socket.bind(&local.into())?; + let bind_again = socket.bind(&local.into()); + socket.connect(&address.into())?; + let connect_again = socket + .connect(&address.into()) + .err() + .and_then(|error| error.raw_os_error()); + Ok((bind_again, connect_again)) + }) + .expect("launcher result") + .expect("workload socket"); + bind_again.expect("repeated bind of the same address"); + assert_eq!(connect_again, Some(libc::EISCONN)); + } + #[test] fn workload_cannot_bind_a_non_loopback_source_address() { // Workload sockets can only present loopback source addresses. A diff --git a/docs/about/support-matrix.mdx b/docs/about/support-matrix.mdx index cfc2fc212c..23d607eb87 100644 --- a/docs/about/support-matrix.mdx +++ b/docs/about/support-matrix.mdx @@ -172,7 +172,6 @@ when it runs inside a container or microVM: | seccomp | Required | Nested user-notification filters and atomic `SECCOMP_IOCTL_NOTIF_ADDFD` with `SECCOMP_ADDFD_FLAG_SEND`, usable under the runtime's existing seccomp profile without added capabilities. The sandbox actively probes these operations before admitting the workload. | | Task-memory access | Required | The non-dumpable broker must be able to read a same-UID, dumpable workload child's memory through `process_vm_readv` or `/proc//mem`. The broker never writes workload memory. The sandbox actively probes the production parent-to-child topology before admitting the workload. | | Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE` before handing it to the workload, without added capabilities (Linux 5.14+). Accepted sockets inherit the binding, so `accept` runs natively and only admits clients in the sandbox's own network namespace. Such a client that is not a workload process may present a non-loopback source address. No broker limit applies to accepted connections; each process's open-file limit (`RLIMIT_NOFILE`), the runtime PID limit, and the sandbox memory limit bound them. The sandbox actively probes that the binding installs, cannot be cleared, and is inherited before admitting the workload. | -| seccomp `WAIT_KILLABLE_RECV` | Optional (Linux 5.19+) | Used when available so a non-fatal signal cannot interrupt a syscall that is waiting for a mediation decision. Mediation behaves the same with or without it, including on RHEL 9.x / RHCOS 5.14. | A kernel version alone does not establish support. A disabled Landlock LSM or a runtime profile that blocks the required seccomp operations causes launch to From f6e8930e4c6b23d4aeba12873805c9fed74f1607 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 18:51:06 -0700 Subject: [PATCH 07/18] fix(sandbox): finish teardown only when every workload descendant has exited Teardown waited only for registered process groups to disappear. A root is unregistered once reaped, so a descendant that ignored SIGTERM could outlive it while termination reported success and never sent SIGKILL, violating the bounded-termination requirement. The sandbox now becomes a child subreaper when it is not PID 1, so orphaned descendants stay in its tree and are reaped. Termination waits until no live descendant remains, and every scanned process is signalled through a pidfd after confirming its start time, so a reused PID is never signalled. Signed-off-by: Drew Newberry --- crates/openshell-sandbox/src/boundary_io.rs | 222 ++++++++++++++---- .../openshell-sandbox/src/boundary_server.rs | 17 +- 2 files changed, 192 insertions(+), 47 deletions(-) diff --git a/crates/openshell-sandbox/src/boundary_io.rs b/crates/openshell-sandbox/src/boundary_io.rs index d404739af0..2a005fd965 100644 --- a/crates/openshell-sandbox/src/boundary_io.rs +++ b/crates/openshell-sandbox/src/boundary_io.rs @@ -220,6 +220,24 @@ impl BoundaryRuntimeState { .is_ok_and(|groups| !groups.is_empty()) } + /// Whether any workload process remains, registered or not. + /// + /// A registered root is unregistered once it is reaped, but descendants + /// that ignored `SIGTERM` may outlive it. When the sandbox owns the + /// process tree (PID 1 or a child subreaper), every live descendant is + /// counted so termination is not reported complete while one survives. + #[must_use] + pub fn has_owned_processes(&self) -> bool { + if self.has_registered_processes() { + return true; + } + #[cfg(target_os = "linux")] + if self.exclusive_pid_namespace && sandbox_owns_process_tree() { + return !owned_processes(&[], true).is_empty(); + } + false + } + /// End the boundary because required standing enforcement was lost. /// /// Returns `true` only to the caller that won the active-to-terminated @@ -265,13 +283,10 @@ impl BoundaryRuntimeState { // requiring ptrace or a capability. let mut previous = Vec::new(); for _ in 0..4 { - let owned = owned_process_ids(&roots, self.exclusive_pid_namespace); - for pid in &owned { - if roots.contains(pid) { - continue; - } - if let Ok(pid) = i32::try_from(*pid) { - let _ = nix::sys::signal::kill(nix::unistd::Pid::from_raw(pid), signal); + let owned = owned_processes(&roots, self.exclusive_pid_namespace); + for process in &owned { + if !roots.contains(&process.pid) { + signal_owned_process(*process, signal); } } if owned == previous { @@ -283,11 +298,51 @@ impl BoundaryRuntimeState { } } +/// One scanned workload process, identified by PID and kernel start time so a +/// reused PID is never mistaken for it. +#[cfg(target_os = "linux")] +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] +struct OwnedProcess { + pid: u32, + start_time: u64, +} + +#[cfg(target_os = "linux")] +struct ProcStat { + parent: u32, + start_time: u64, + live: bool, +} + #[cfg(target_os = "linux")] -fn owned_process_ids(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { - let mut parents = HashMap::new(); +fn read_proc_stat(pid: u32) -> Option { + let stat = std::fs::read_to_string(format!("/proc/{pid}/stat")).ok()?; + // The command name may contain spaces or parentheses; fields resume after + // the final ") ". Field 3 is the state, 4 the parent, 22 the start time. + let fields = stat + .rsplit_once(") ")? + .1 + .split_whitespace() + .collect::>(); + Some(ProcStat { + parent: fields.get(1)?.parse().ok()?, + start_time: fields.get(19)?.parse().ok()?, + live: !matches!(*fields.first()?, "Z" | "X" | "x"), + }) +} + +/// Whether orphaned descendants are reparented to this sandbox process. +#[cfg(target_os = "linux")] +fn sandbox_owns_process_tree() -> bool { + std::process::id() == 1 + || rustix::process::child_subreaper().is_ok_and(|subreaper| subreaper.is_some()) +} + +#[cfg(target_os = "linux")] +fn owned_processes(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { + let mut stats = HashMap::new(); let Ok(entries) = std::fs::read_dir("/proc") else { - return roots.to_vec(); + return Vec::new(); }; for entry in entries.flatten() { let Some(pid) = entry @@ -297,41 +352,25 @@ fn owned_process_ids(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { else { continue; }; - let Ok(stat) = std::fs::read_to_string(entry.path().join("stat")) else { - continue; - }; - let Some(after_name) = stat.rsplit_once(") ").map(|(_, fields)| fields) else { - continue; - }; - let Some(parent) = after_name - .split_whitespace() - .nth(1) - .and_then(|field| field.parse::().ok()) - else { - continue; - }; - parents.insert(pid, parent); - } - - // When openshell-sandbox is PID 1, every other process in its exclusive - // namespace is workload-owned, including an orphan reparented during the - // scan. Outside that deployment shape, restrict the walk to registered - // roots so unit tests and development runs cannot affect sibling tasks. - if exclusive_pid_namespace && std::process::id() == 1 { - let mut owned = parents - .keys() - .copied() - .filter(|pid| *pid != 1) - .collect::>(); - owned.sort_unstable(); - return owned; + if let Some(stat) = read_proc_stat(pid) { + stats.insert(pid, stat); + } } - let mut owned = roots.to_vec(); + // When the sandbox is PID 1 of its exclusive namespace or a child + // subreaper, orphans are reparented to it, so its descendants are exactly + // the workload tree, including an orphan reparented during the scan. + // Otherwise restrict the walk to registered roots so unit tests and + // development runs cannot affect sibling tasks. + let mut owned = if exclusive_pid_namespace && sandbox_owns_process_tree() { + vec![std::process::id()] + } else { + roots.to_vec() + }; loop { let mut changed = false; - for (&pid, &parent) in &parents { - if !owned.contains(&pid) && owned.contains(&parent) { + for (&pid, stat) in &stats { + if !owned.contains(&pid) && owned.contains(&stat.parent) { owned.push(pid); changed = true; } @@ -340,11 +379,45 @@ fn owned_process_ids(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { break; } } + let sandbox = std::process::id(); + let mut owned = owned + .into_iter() + .filter(|pid| *pid != sandbox) + .filter_map(|pid| { + let stat = stats.get(&pid)?; + stat.live.then_some(OwnedProcess { + pid, + start_time: stat.start_time, + }) + }) + .collect::>(); owned.sort_unstable(); owned.dedup(); owned } +/// Signal one scanned process through a pidfd, after confirming the pidfd +/// refers to the scanned process rather than a later process with its PID. +#[cfg(target_os = "linux")] +fn signal_owned_process(process: OwnedProcess, signal: nix::sys::signal::Signal) { + let Some(pid) = i32::try_from(process.pid) + .ok() + .and_then(rustix::process::Pid::from_raw) + else { + return; + }; + let Some(signal) = rustix::process::Signal::from_named_raw(signal as i32) else { + return; + }; + let Ok(pidfd) = rustix::process::pidfd_open(pid, rustix::process::PidfdFlags::empty()) else { + return; + }; + if read_proc_stat(process.pid).map(|stat| stat.start_time) != Some(process.start_time) { + return; + } + let _ = rustix::process::pidfd_send_signal(&pidfd, signal); +} + #[derive(Clone)] struct RegisteredProcessGroup { pid: u32, @@ -509,6 +582,73 @@ mod tests { )); } + #[cfg(target_os = "linux")] + #[test] + fn termination_waits_for_descendants_that_outlive_their_root() { + use std::io::BufRead as _; + use std::os::unix::process::CommandExt as _; + + // Becoming a subreaper changes this whole process, so run the + // scenario in a fresh copy of the test binary. + const CHILD_MARKER: &str = "OPENSHELL_SUBREAPER_TEARDOWN_CHILD"; + if std::env::var_os(CHILD_MARKER).is_none() { + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "boundary_io::tests::termination_waits_for_descendants_that_outlive_their_root", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + .expect("run isolated teardown test"); + assert!(status.success(), "isolated teardown test failed"); + return; + } + rustix::process::set_child_subreaper(Some(rustix::process::getpid())) + .expect("become child subreaper"); + let runtime = BoundaryRuntimeState::new_exclusive_pid_namespace(); + // The grandchild inherits an ignored SIGTERM; the root restores the + // default disposition and exits on SIGTERM. + let mut root = std::process::Command::new("/bin/sh") + .args([ + "-c", + "trap '' TERM; sleep 600 & trap - TERM; echo ready; wait", + ]) + .process_group(0) + .stdout(std::process::Stdio::piped()) + .spawn() + .expect("spawn root"); + let mut line = String::new(); + std::io::BufReader::new(root.stdout.take().unwrap()) + .read_line(&mut line) + .unwrap(); + assert_eq!(line.trim(), "ready"); + let terminal = Arc::new(std::sync::atomic::AtomicBool::new(false)); + runtime + .register_process_group(root.id(), terminal.clone(), Arc::new(Mutex::new(()))) + .expect("register root"); + + assert!(runtime.begin_termination()); + root.wait().expect("root exits on SIGTERM"); + terminal.store(true, Ordering::Release); + runtime.unregister_process_group(root.id(), &terminal); + assert!(!runtime.has_registered_processes()); + assert!( + runtime.has_owned_processes(), + "a SIGTERM-ignoring grandchild must keep termination incomplete" + ); + + runtime.force_kill(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(5); + while runtime.has_owned_processes() { + assert!( + std::time::Instant::now() < deadline, + "forced termination left a descendant alive" + ); + std::thread::sleep(std::time::Duration::from_millis(20)); + } + } + #[test] fn freeze_blocks_new_operations_until_explicit_resume() { let runtime = BoundaryRuntimeState::new_exclusive_pid_namespace(); diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index 4695d65b31..fbf2d2d471 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -222,10 +222,15 @@ mod linux { } crate::sandbox::apply_supervisor_startup_hardening() .map_err(|error| format!("install sandbox process prelude: {error}"))?; - if nix::unistd::getpid().as_raw() == 1 { - crate::managed_children::start_orphan_reaper() - .map_err(|error| format!("start sandbox orphan reaper: {error}"))?; - } + // Keep orphaned workload descendants in this process tree so + // termination can find and kill them, then reap the adopted ones. + // PID 1 already receives orphans; elsewhere become a child subreaper. + if nix::unistd::getpid().as_raw() != 1 { + rustix::process::set_child_subreaper(Some(rustix::process::getpid())) + .map_err(|error| format!("become child subreaper: {error}"))?; + } + crate::managed_children::start_orphan_reaper() + .map_err(|error| format!("start sandbox orphan reaper: {error}"))?; let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() .map_err(|error| format!("start sandbox workload launcher: {error}"))?; let protected_control_port = match &config.listener { @@ -1893,12 +1898,12 @@ mod linux { async fn wait_for_process_tree_exit(process: &ManagedProcess, timeout: Duration) -> bool { let deadline = tokio::time::Instant::now() + timeout; - while process.boundary_runtime.has_registered_processes() + while process.boundary_runtime.has_owned_processes() && tokio::time::Instant::now() < deadline { tokio::time::sleep(Duration::from_millis(25)).await; } - !process.boundary_runtime.has_registered_processes() + !process.boundary_runtime.has_owned_processes() } fn shutdown(&self) { From ade074fe6178e08eaf299e6ad6c46691d0e564fb Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 18:56:46 -0700 Subject: [PATCH 08/18] fix(sandbox): keep a frozen workload from resuming itself Freezing stops every workload process with SIGSTOP. tgkill and rt_tgsigqueueinfo were not mediated, and mediated kill, rt_sigqueueinfo, and tkill passed SIGCONT through, so a workload process that was not yet stopped could resume the others while the supervisor recovered. Mediate tgkill and rt_tgsigqueueinfo like tkill: refuse targets in the sandbox thread group, report a thread outside the named group as missing, and continue otherwise. The boundary marks the broker frozen before stopping the workload and clears it after resuming; while frozen, workload requests to send SIGCONT fail with EPERM. Signed-off-by: Drew Newberry --- .../src/linux/process_signal.rs | 47 +++++++++-- .../src/linux/seccomp_notify.rs | 2 + .../openshell-sandbox/src/boundary_server.rs | 2 + .../openshell-sandbox/src/network_broker.rs | 82 ++++++++++++++++++- 4 files changed, 123 insertions(+), 10 deletions(-) diff --git a/crates/openshell-isolation-interface/src/linux/process_signal.rs b/crates/openshell-isolation-interface/src/linux/process_signal.rs index 796f548037..67fcae68b1 100644 --- a/crates/openshell-isolation-interface/src/linux/process_signal.rs +++ b/crates/openshell-isolation-interface/src/linux/process_signal.rs @@ -27,6 +27,7 @@ pub fn mediate_process_signal( listener: &NotificationListener, notification: Notification, sandbox_tgid: u32, + workload_frozen: bool, ) -> io::Result<()> { listener.validate_id(notification.id)?; let target = scalar_int(notification.args[0]); @@ -38,6 +39,7 @@ pub fn mediate_process_signal( libc::EINVAL })); } + refuse_resume_while_frozen(signal, workload_frozen)?; let target = u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; let retained = retain_signal_target(target, sandbox_tgid)?; // SAFETY: all-zero siginfo consists of valid integer/pointer fields. A @@ -92,26 +94,53 @@ pub fn mediate_thread_signal( listener: &NotificationListener, notification: Notification, sandbox_tgid: u32, + workload_frozen: bool, ) -> io::Result<()> { listener.validate_id(notification.id)?; - let target = scalar_int(notification.args[0]); - let signal = scalar_int(notification.args[1]); - if target <= 0 || !(0..=64).contains(&signal) { - return Err(io::Error::from_raw_os_error(if target <= 0 { - libc::EPERM - } else { - libc::EINVAL - })); + // tkill(tid, sig); tgkill(tgid, tid, sig); rt_tgsigqueueinfo(tgid, tid, + // sig, info). The kernel itself validates a queued siginfo's code. + let (claimed_group, target, signal) = match i64::from(notification.syscall) { + libc::SYS_tkill => ( + None, + scalar_int(notification.args[0]), + scalar_int(notification.args[1]), + ), + libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo => ( + Some(scalar_int(notification.args[0])), + scalar_int(notification.args[1]), + scalar_int(notification.args[2]), + ), + _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), + }; + if target <= 0 || claimed_group.is_some_and(|group| group <= 0) { + return Err(io::Error::from_raw_os_error(libc::EPERM)); } + if !(0..=64).contains(&signal) { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); + } + refuse_resume_while_frozen(signal, workload_frozen)?; let target = u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; let target_group = thread_group_id(target)?; if target_group == sandbox_tgid || target_group == 0 { return Err(io::Error::from_raw_os_error(libc::EPERM)); } + if claimed_group.is_some_and(|group| u32::try_from(group).ok() != Some(target_group)) { + // The kernel reports a thread outside the named group as missing. + return Err(io::Error::from_raw_os_error(libc::ESRCH)); + } listener.validate_id(notification.id)?; listener.respond_continue(notification.id) } +/// While the boundary has stopped the workload for supervisor recovery, a +/// workload process that was not yet stopped must not resume the others. +fn refuse_resume_while_frozen(signal: i32, workload_frozen: bool) -> io::Result<()> { + if workload_frozen && signal == libc::SIGCONT { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } + Ok(()) +} + fn scalar_int(value: u64) -> i32 { let bytes = value.to_ne_bytes(); #[cfg(target_endian = "little")] @@ -177,7 +206,7 @@ mod tests { .unwrap(); let notification = listener.receive().unwrap(); let error = - mediate_process_signal(&listener, notification, std::process::id()).unwrap_err(); + mediate_process_signal(&listener, notification, std::process::id(), false).unwrap_err(); assert_eq!(error.raw_os_error(), Some(libc::EPERM)); listener .respond_errno(notification.id, libc::EPERM) diff --git a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs index 6cf48abd4d..0e447bc817 100644 --- a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs +++ b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs @@ -342,6 +342,8 @@ pub fn install_workload_listener() -> io::Result { libc::SYS_setsockopt, libc::SYS_kill, libc::SYS_tkill, + libc::SYS_tgkill, + libc::SYS_rt_tgsigqueueinfo, libc::SYS_rt_sigqueueinfo, libc::SYS_openat, libc::SYS_openat2, diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index fbf2d2d471..db825c6292 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -1716,6 +1716,7 @@ mod linux { "frozen workload could not be resumed".to_string(), )); } + self.network_broker.set_workload_frozen(false); tracing::info!( connection_id = ?principal.connection_id(), "Sandbox Protocol connection recovered; workload resumed" @@ -1772,6 +1773,7 @@ mod linux { return; } if let Some(process) = &process { + self.network_broker.set_workload_frozen(true); let _ = process.boundary_runtime.freeze(); } *connection = SupervisorConnectionState::Frozen { recovery_id }; diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 93ed821966..ec0d11e2b4 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -168,6 +168,7 @@ fn register_dns_socket( #[derive(Clone)] struct NotificationQueues { provider_files: crate::provider_files::ProviderFiles, + workload_frozen: Arc, protected_control_port: Option, identity_resolver: ProcfsIdentityResolver, pending: mpsc::Sender, @@ -181,6 +182,7 @@ struct NotificationQueues { #[derive(Clone)] pub struct NetworkBroker { provider_files: crate::provider_files::ProviderFiles, + workload_frozen: Arc, pending: Arc>>, pending_dns: Arc>>, dns_address: SocketAddr, @@ -232,8 +234,10 @@ impl NetworkBroker { let retained_socket_capacity = retained_socket_capacity()?; let registry = Arc::new(Mutex::new(SocketRegistry::new(1, SOCKET_CAPACITY)?)); let provider_files = crate::provider_files::ProviderFiles::default(); + let workload_frozen = Arc::new(AtomicBool::new(false)); let queues = NotificationQueues { provider_files: provider_files.clone(), + workload_frozen: workload_frozen.clone(), protected_control_port, identity_resolver: ProcfsIdentityResolver::for_pid_namespace(), pending: pending_tx, @@ -283,6 +287,7 @@ impl NetworkBroker { .map_err(|error| io::Error::other(format!("start network broker: {error}")))?; Ok(Self { provider_files, + workload_frozen, pending: Arc::new(tokio::sync::Mutex::new(pending_rx)), pending_dns: Arc::new(tokio::sync::Mutex::new(pending_dns_rx)), dns_address, @@ -290,6 +295,14 @@ impl NetworkBroker { }) } + /// Record whether the boundary has stopped the workload for supervisor + /// recovery. While frozen, workload requests to send `SIGCONT` are refused + /// so a process that was not yet stopped cannot resume the others. Set + /// this before stopping the workload and clear it after resuming it. + pub(crate) fn set_workload_frozen(&self, frozen: bool) { + self.workload_frozen.store(frozen, Ordering::Release); + } + pub(crate) async fn accept(&self) -> io::Result { self.pending .lock() @@ -515,13 +528,18 @@ fn dispatch_notification( &listener, notification, std::process::id(), + queues.workload_frozen.load(Ordering::Acquire), ); } - if syscall == libc::SYS_tkill { + if matches!( + syscall, + libc::SYS_tkill | libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo + ) { return openshell_isolation_interface::linux::process_signal::mediate_thread_signal( &listener, notification, std::process::id(), + queues.workload_frozen.load(Ordering::Acquire), ); } if syscall == libc::SYS_socket { @@ -2390,6 +2408,68 @@ mod tests { assert_eq!(connect_again, Some(libc::EISCONN)); } + #[test] + fn frozen_workload_cannot_resume_processes_with_sigcont() { + // A workload process that is not yet stopped when the boundary + // freezes must not resume the others, through process-directed + // (kill) or thread-directed (tgkill) signals. + const CHILD_MARKER: &str = "OPENSHELL_FROZEN_SIGCONT_CHILD"; + if std::env::var_os(CHILD_MARKER).is_some() { + let errno = |result: nix::Result<()>| result.err().map_or(0, |error| error as i32); + let kill = errno(nix::sys::signal::kill( + nix::unistd::getpid(), + nix::sys::signal::Signal::SIGCONT, + )); + // SAFETY: tgkill takes scalar arguments naming this thread. + let tgkill = unsafe { + libc::syscall( + libc::SYS_tgkill, + libc::getpid(), + libc::gettid(), + libc::SIGCONT, + ) + }; + let tgkill = if tgkill < 0 { + io::Error::last_os_error().raw_os_error().unwrap_or(-1) + } else { + 0 + }; + println!("kill={kill} tgkill={tgkill}"); + return; + } + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let run_workload = |launcher: &openshell_isolation_interface::linux::workload_launcher::WorkloadLauncher| { + let output = launcher + .execute(|| { + std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "network_broker::tests::frozen_workload_cannot_resume_processes_with_sigcont", + "--nocapture", + "--quiet", + ]) + .env(CHILD_MARKER, "1") + .output() + }) + .unwrap() + .expect("run workload child"); + String::from_utf8_lossy(&output.stdout) + .lines() + .find(|line| line.starts_with("kill=")) + .expect("workload child result") + .to_string() + }; + broker.set_workload_frozen(true); + assert_eq!( + run_workload(&launcher), + format!("kill={} tgkill={}", libc::EPERM, libc::EPERM) + ); + broker.set_workload_frozen(false); + assert_eq!(run_workload(&launcher), "kill=0 tgkill=0"); + } + #[test] fn workload_cannot_bind_a_non_loopback_source_address() { // Workload sockets can only present loopback source addresses. A From f5615df5a39eea65537112283912cf76c13e0f9f Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 19:07:13 -0700 Subject: [PATCH 09/18] fix(sandbox): send mediated DNS datagrams from the broker socket The native UDP send path re-ran the workload's own sendmsg, so per-message ancillary data such as IP_PKTINFO rode along; only the loopback destination contained a routing override. The broker now reads the datagram and sends it from its retained, relay-connected socket, which it builds without ancillary data, so a per-message override cannot redirect the packet. It writes nothing back into workload memory. A send carrying control data is refused with EOPNOTSUPP rather than silently stripped. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/network_broker.rs | 165 ++++++++++++++++-- 1 file changed, 147 insertions(+), 18 deletions(-) diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index ec0d11e2b4..2e5899b983 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -1170,6 +1170,8 @@ fn classify_send( libc::SYS_sendmsg => vec![read_sendmsg_message( notification.tid, notification.args[1], + i32::try_from(notification.args[2]) + .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, )?], libc::SYS_sendmmsg => read_sendmmsg_messages(notification)?, _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), @@ -1182,11 +1184,15 @@ fn classify_send( if entry.metadata().kind == InetKind::DnsUdp && matches!(entry.state(), SocketState::Local { .. }) => { - if messages.iter().all(|message| message.destination.is_none()) { - listener.respond_continue(notification.id) - } else { - Err(io::Error::from_raw_os_error(libc::EACCES)) + if !messages.iter().all(|message| message.destination.is_none()) { + return Err(io::Error::from_raw_os_error(libc::EACCES)); + } + let source_fd = entry.retained_preconnect()?.as_raw_fd(); + listener.validate_id(notification.id)?; + for message in &messages { + send_dns_message(source_fd, message)?; } + listener.respond_value(notification.id, dns_send_result(syscall, &messages)) } Ok(entry) if matches!(entry.state(), SocketState::DnsUdp { .. }) => { let SocketState::DnsUdp { relay } = entry.state() else { @@ -1199,15 +1205,20 @@ fn classify_send( // destination is absent or names that same relay. The mandatory // outer network fence remains the fail-closed backstop for the // sibling-thread pointer race inherent in seccomp CONTINUE. - if messages.iter().all(|message| { + let relay = *relay; + if !messages.iter().all(|message| { message .destination - .is_none_or(|destination| destination == *relay) + .is_none_or(|destination| destination == relay) }) { - listener.respond_continue(notification.id) - } else { - Err(io::Error::from_raw_os_error(libc::EACCES)) + return Err(io::Error::from_raw_os_error(libc::EACCES)); } + let source_fd = entry.retained_preconnect()?.as_raw_fd(); + listener.validate_id(notification.id)?; + for message in &messages { + send_dns_message(source_fd, message)?; + } + listener.respond_value(notification.id, dns_send_result(syscall, &messages)) } Ok(entry) if entry.metadata().kind == InetKind::DnsUdp @@ -1230,14 +1241,15 @@ fn classify_send( lock(&dns_relay.udp_admissions).remove(&peer); return Err(error); } + for message in &messages { + send_dns_message(source_fd, message)?; + } + // Keep the broker's retained socket so later sends also originate + // from it; the workload's aliased fd shares this open file. entry.set_state(SocketState::DnsUdp { relay: dns_relay.address, }); - entry.release_preconnect(); - // The socket is now pinned to the relay and bound to loopback. The - // kernel performs the send and writes any per-message results, so - // the broker never writes workload memory. - listener.respond_continue(notification.id) + listener.respond_value(notification.id, dns_send_result(syscall, &messages)) } Ok(_) => Err(io::Error::from_raw_os_error(libc::EDESTADDRREQ)), // Non-INET sockets and natively accepted sockets were never @@ -1264,10 +1276,19 @@ fn send_flags(syscall: i64, args: [u64; 6]) -> i32 { } struct SendMessage { + data: Vec, destination: Option, + flags: i32, } fn read_sendto_message(notification: Notification) -> io::Result { + let length = usize::try_from(notification.args[2]) + .map_err(|_| io::Error::from_raw_os_error(libc::EMSGSIZE))?; + if u16::try_from(length).is_err() { + return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); + } + let mut data = vec![0_u8; length]; + task_memory::read_exact(notification.tid, notification.args[1], &mut data)?; let destination = if notification.args[4] == 0 { None } else { @@ -1277,11 +1298,19 @@ fn read_sendto_message(notification: Notification) -> io::Result { notification.args[5], )?) }; - Ok(SendMessage { destination }) + Ok(SendMessage { + data, + destination, + flags: i32::try_from(notification.args[3]) + .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, + }) } -fn read_sendmsg_message(tid: u32, address: u64) -> io::Result { +fn read_sendmsg_message(tid: u32, address: u64, flags: i32) -> io::Result { let header = read_task_value::(tid, address)?; + // Ancillary data can carry per-message routing overrides (IP_PKTINFO). + // The broker sends from its own socket without control messages, so a + // workload that needs them is refused rather than silently stripped. if header.msg_controllen != 0 { return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); } @@ -1294,7 +1323,38 @@ fn read_sendmsg_message(tid: u32, address: u64) -> io::Result { u64::from(header.msg_namelen), )?) }; - Ok(SendMessage { destination }) + #[cfg(target_env = "musl")] + let iov_count = usize::try_from(header.msg_iovlen) + .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; + #[cfg(not(target_env = "musl"))] + let iov_count = header.msg_iovlen; + if iov_count > 32 { + return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); + } + let mut data = Vec::new(); + for index in 0..iov_count { + let offset = index + .checked_mul(size_of::()) + .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?; + let iov = read_task_value::( + tid, + (header.msg_iov as u64) + .checked_add(u64::try_from(offset).unwrap_or(u64::MAX)) + .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?, + )?; + let start = data.len(); + let end = start + .checked_add(iov.iov_len) + .filter(|length| u16::try_from(*length).is_ok()) + .ok_or_else(|| io::Error::from_raw_os_error(libc::EMSGSIZE))?; + data.resize(end, 0); + task_memory::read_exact(tid, iov.iov_base as u64, &mut data[start..end])?; + } + Ok(SendMessage { + data, + destination, + flags, + }) } fn read_sendmmsg_messages(notification: Notification) -> io::Result> { @@ -1303,6 +1363,8 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result 32 { return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); } + let flags = i32::try_from(notification.args[3]) + .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; (0..count) .map(|index| { let offset = index @@ -1311,11 +1373,39 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result io::Result<()> { + // Strip MSG_FASTOPEN (already rejected) and MSG_MORE (no corking here). + let flags = message.flags & !(libc::MSG_FASTOPEN | libc::MSG_MORE); + // SAFETY: `fd` is the retained relay-connected UDP socket; the buffer is + // live for the duration of the call. + let sent = unsafe { libc::send(fd, message.data.as_ptr().cast(), message.data.len(), flags) }; + if sent < 0 { + return Err(io::Error::last_os_error()); + } + if usize::try_from(sent).ok() == Some(message.data.len()) { + Ok(()) + } else { + Err(io::Error::from_raw_os_error(libc::EIO)) + } +} + +fn dns_send_result(syscall: i64, messages: &[SendMessage]) -> i64 { + if syscall == libc::SYS_sendmmsg { + i64::try_from(messages.len()).unwrap_or(i64::MAX) + } else { + i64::try_from(messages.first().map_or(0, |message| message.data.len())).unwrap_or(i64::MAX) + } +} + fn read_task_value(tid: u32, address: u64) -> io::Result { let mut bytes = vec![0_u8; size_of::()]; task_memory::read_exact(tid, address, &mut bytes)?; @@ -2627,6 +2717,45 @@ mod tests { ); } + #[test] + fn dns_send_with_ancillary_data_is_refused() { + // Ancillary control data (e.g. IP_PKTINFO) can carry a per-message + // routing override. The broker sends from its own socket without + // control messages, so a workload that supplies them is refused. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let dns_address = broker.dns_address(); + let errno = launcher + .execute(move || { + let socket = UdpSocket::bind("0.0.0.0:0").unwrap(); + let native = socket2::SockAddr::from(dns_address); + let payload = *b"dns-query"; + let mut iov = libc::iovec { + iov_base: payload.as_ptr().cast_mut().cast(), + iov_len: payload.len(), + }; + // One IP_PKTINFO control message. + let mut control = [0_u8; 32]; + let header = libc::msghdr { + msg_name: native.as_ptr().cast_mut().cast(), + msg_namelen: native.len(), + msg_iov: &raw mut iov, + msg_iovlen: 1, + msg_control: control.as_mut_ptr().cast(), + msg_controllen: control.len(), + msg_flags: 0, + }; + // SAFETY: the header references live local buffers for the call. + let sent = unsafe { libc::sendmsg(socket.as_raw_fd(), &raw const header, 0) }; + (sent < 0) + .then(|| io::Error::last_os_error().raw_os_error()) + .flatten() + }) + .expect("launcher result"); + assert_eq!(errno, Some(libc::EOPNOTSUPP)); + } + #[test] fn udp_dns_allows_repeated_destination_sends_to_the_pinned_relay() { let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() From c4dea19adb139f721d5f0295d46f43d5c3bcfee0 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 19:10:38 -0700 Subject: [PATCH 10/18] fix(sandbox): keep sandbox control variables out of the workload environment The canonical process inherits the sandbox environment so the image's own ENV reaches the workload, then removed only a denylist of credential variables. Other variables in the reserved OPENSHELL_ namespace, such as the serialized user environment and the log level, still reached the workload. Remove every inherited OPENSHELL_ variable before applying the declared environment, and restore OPENSHELL_SANDBOX=1. The image's ordinary ENV and declared variables are unaffected; declared variables cannot use the reserved namespace. Signed-off-by: Drew Newberry --- crates/openshell-sandbox/src/process.rs | 85 +++++++++++++++++++++++++ 1 file changed, 85 insertions(+) diff --git a/crates/openshell-sandbox/src/process.rs b/crates/openshell-sandbox/src/process.rs index 595ceac29e..20ae60d5b6 100644 --- a/crates/openshell-sandbox/src/process.rs +++ b/crates/openshell-sandbox/src/process.rs @@ -105,6 +105,9 @@ pub(crate) fn ca_runtime_read_only_paths(ca_paths: Option<&(PathBuf, PathBuf)>) paths } +/// Prefix of environment variable names reserved for `OpenShell`. +const RESERVED_ENV_PREFIX: &str = "OPENSHELL_"; + const SUPERVISOR_ONLY_ENV_VARS: &[&str] = &[ openshell_core::sandbox_env::OCI_IMAGE_USER, openshell_core::sandbox_env::SANDBOX_UID, @@ -205,6 +208,21 @@ fn apply_canonical_process_environment( interactive: bool, user_environment: &HashMap, ) { + // The canonical process inherits the sandbox's environment so the image's + // own ENV (PATH, LANG, JAVA_HOME, ...) reaches the workload. Remove the + // reserved OPENSHELL_ namespace inherited from the sandbox itself, which + // carries its own control state (for example the serialized user + // environment and log level), then restore the one marker the workload is + // meant to see. Declared variables cannot use this namespace. + for (key, _) in std::env::vars_os() { + if key + .to_str() + .is_some_and(|key| key.starts_with(RESERVED_ENV_PREFIX)) + { + cmd.env_remove(key); + } + } + cmd.env(openshell_core::sandbox_env::SANDBOX, "1"); cmd.envs(user_environment); let (session_user, session_home) = session_user_and_home(policy, workspace.home()); // Resolve a shell present in the workload image. This code runs inside the @@ -1122,6 +1140,73 @@ mod tests { assert_eq!(variables.get("TERM"), Some(&"xterm-256color")); } + #[cfg(unix)] + #[test] + fn canonical_process_drops_inherited_reserved_environment() { + // The sandbox's own control variables live in the reserved + // OPENSHELL_ namespace and must not reach the workload, while the + // image's ordinary ENV must. Run in a fresh copy of the test binary + // so the test harness environment is untouched. + const CHILD_MARKER: &str = "OPENSHELL_TEST_RESERVED_ENV_CHILD"; + if std::env::var_os(CHILD_MARKER).is_none() { + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "process::tests::canonical_process_drops_inherited_reserved_environment", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .env(openshell_core::sandbox_env::LOG_LEVEL, "debug") + .env(openshell_core::sandbox_env::USER_ENVIRONMENT, "{}") + .env("IMAGE_LANG", "keep") + .status() + .expect("run isolated environment test"); + assert!(status.success(), "isolated environment test failed"); + return; + } + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .unwrap(); + let current_user = User::from_uid(nix::unistd::geteuid()).unwrap().unwrap(); + let policy = policy_with_process(ProcessPolicy { + run_as_user: Some(current_user.name), + run_as_group: None, + }); + // Mirror production: inherit the sandbox environment, no env_clear. + let mut cmd = Command::new("/usr/bin/env"); + cmd.stdout(StdStdio::piped()); + apply_canonical_process_environment( + &mut cmd, + &policy, + &ResolvedWorkspace::default(), + false, + &HashMap::from([("DECLARED".into(), "yes".into())]), + ); + let output = runtime + .block_on(async { cmd.output().await }) + .expect("run environment probe"); + assert!(output.status.success()); + let environment = String::from_utf8(output.stdout).unwrap(); + let variables: HashMap<_, _> = environment + .lines() + .filter_map(|line| line.split_once('=')) + .collect(); + assert!( + !variables + .keys() + .any(|key| key.starts_with(RESERVED_ENV_PREFIX) + && *key != openshell_core::sandbox_env::SANDBOX), + "reserved variables reached the workload: {variables:?}" + ); + assert_eq!( + variables.get(openshell_core::sandbox_env::SANDBOX), + Some(&"1") + ); + assert_eq!(variables.get("IMAGE_LANG"), Some(&"keep")); + assert_eq!(variables.get("DECLARED"), Some(&"yes")); + } + #[cfg(unix)] #[tokio::test] async fn canonical_process_receives_declared_environment_and_home() { From adfa02b7975f68e4e8a27cdd81aa5f3c4bad1c13 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 19:15:25 -0700 Subject: [PATCH 11/18] fix(sandbox): keep /proc read-only for GPU workloads GPU mode granted read-write access to all of /proc so CUDA's cuInit could write thread names to /proc//task//comm. Keep /proc read-only. Open mediation now serves a writable open of the caller's own thread comm file: the broker opens it and injects the descriptor, with no syscall continued. The kernel accepts a comm write only from the target's own thread group, so a substituted path or reused thread ID cannot rename another process's thread. Any other /proc write is left to Landlock, which denies it. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/boundary_server.rs | 15 +- .../openshell-sandbox/src/provider_files.rs | 170 +++++++++++++++++- 2 files changed, 172 insertions(+), 13 deletions(-) diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index db825c6292..5593627653 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -85,16 +85,15 @@ mod linux { // NVML may traverse the persistenced socket directory during initialization; // WSL2 supplies GPU libraries under /usr/lib/wsl and the /dev/dxg device. const GPU_BASELINE_READ_ONLY: &[&str] = &["/run/nvidia-persistenced", "/usr/lib/wsl"]; - // CUDA opens device nodes read-write and writes thread names through - // /proc//task//comm during cuInit(). A /proc/self rule would bind - // to the launcher's inodes, not those of its workload children. + // CUDA opens device nodes read-write. Its thread-name writes through + // /proc//task//comm are served by open mediation, so /proc + // stays read-only. const GPU_BASELINE_READ_WRITE: &[&str] = &[ "/dev/nvidiactl", "/dev/nvidia-uvm", "/dev/nvidia-uvm-tools", "/dev/nvidia-modeset", "/dev/dxg", - "/proc", ]; fn duration_micros(duration: Duration) -> u64 { @@ -153,13 +152,7 @@ mod linux { continue; } if policy.filesystem.read_only.contains(&path) { - if path != Path::new("/proc") { - continue; - } - policy - .filesystem - .read_only - .retain(|allowed| allowed != &path); + continue; } policy.filesystem.read_write.push(path); modified = true; diff --git a/crates/openshell-sandbox/src/provider_files.rs b/crates/openshell-sandbox/src/provider_files.rs index ab719cd355..e19134a74f 100644 --- a/crates/openshell-sandbox/src/provider_files.rs +++ b/crates/openshell-sandbox/src/provider_files.rs @@ -17,6 +17,7 @@ use openshell_isolation_interface::linux::seccomp_notify::{Notification, Notific use openshell_isolation_interface::linux::task_memory; const PREFIX: &str = "/run/openshell/providers/"; +const PROC_PREFIX: &str = "/proc/"; const MAX_FILE_BYTES: usize = 65_536; const MAX_TOTAL_BYTES: usize = 262_144; const MAX_PATH_BYTES: usize = 4_096; @@ -75,6 +76,9 @@ impl ProviderFiles { } else { notification.args[0] }; + if handle_thread_comm_open(listener, notification, path_address)? { + return Ok(()); + } // Every workload open reaches the listener. Copy only the reserved // prefix for ordinary paths; full path reads are rare. let mut prefix = [0_u8; PREFIX.len()]; @@ -160,6 +164,99 @@ fn validate_path(path: &str) -> io::Result<()> { Ok(()) } +/// Serve a write open of the caller's own thread name file. +/// +/// `pthread_setname_np` and CUDA's `cuInit` rename threads by writing +/// `/proc//task//comm`. Landlock keeps `/proc` read-only, so the +/// broker opens the caller's own `comm` file and injects the descriptor; no +/// syscall is continued. The kernel's `comm_write` accepts a write only from +/// the target's own thread group, so a substituted path or reused thread ID +/// cannot rename another process's thread through the descriptor. Returns +/// `false` when the open is not such a request and normal mediation applies. +fn handle_thread_comm_open( + listener: &NotificationListener, + notification: Notification, + path_address: u64, +) -> io::Result { + let Ok(flags) = open_flags(¬ification) else { + return Ok(false); + }; + let access = flags & libc::O_ACCMODE; + // Reads are already allowed by the read-only /proc rule. + if access == libc::O_RDONLY { + return Ok(false); + } + let mut prefix = [0_u8; PROC_PREFIX.len()]; + if task_memory::read_exact(notification.tid, path_address, &mut prefix).is_err() + || prefix != PROC_PREFIX.as_bytes() + { + return Ok(false); + } + let Ok(path) = read_path(notification.tid, path_address) else { + return Ok(false); + }; + let Some(caller_group) = thread_group_of(notification.tid) else { + return Ok(false); + }; + let Some(target) = comm_target(&path, notification.tid, caller_group) else { + return Ok(false); + }; + // Anything else, including another process's thread, is left to + // Landlock, which denies the write. + if thread_group_of(target) != Some(caller_group) { + return Ok(false); + } + if flags + & (libc::O_CREAT + | libc::O_EXCL + | libc::O_TRUNC + | libc::O_TMPFILE + | libc::O_DIRECTORY + | libc::O_PATH) + != 0 + { + listener.respond_errno(notification.id, libc::EINVAL)?; + return Ok(true); + } + let file = std::fs::OpenOptions::new() + .read(access == libc::O_RDWR) + .write(true) + .open(format!("/proc/{caller_group}/task/{target}/comm"))?; + listener.add_fd_and_send( + notification.id, + file.as_raw_fd(), + flags & libc::O_CLOEXEC != 0, + )?; + Ok(true) +} + +/// Resolve the thread whose `comm` file `path` names, if it is the caller's +/// own thread group. +fn comm_target(path: &str, caller_tid: u32, caller_group: u32) -> Option { + let parts = path + .strip_prefix(PROC_PREFIX)? + .split('/') + .collect::>(); + let own_group = |part: &str| part == "self" || part.parse::().ok() == Some(caller_group); + match parts.as_slice() { + ["thread-self", "comm"] => Some(caller_tid), + [group, "comm"] if own_group(group) => Some(caller_group), + [group, "task", tid, "comm"] if own_group(group) => tid.parse().ok(), + _ => None, + } +} + +/// Thread group (process) ID of a thread, from its procfs status. +fn thread_group_of(tid: u32) -> Option { + std::fs::read_to_string(format!("/proc/{tid}/status")) + .ok()? + .lines() + .find_map(|line| line.strip_prefix("Tgid:"))? + .trim() + .parse() + .ok() +} + fn read_path(tid: u32, mut address: u64) -> io::Result { if address == 0 { return Err(io::Error::from_raw_os_error(libc::EFAULT)); @@ -228,12 +325,81 @@ fn sealed_memfd(content: &[u8]) -> io::Result { #[cfg(test)] mod tests { - use super::{ProviderFiles, sealed_memfd}; + use super::{ProviderFiles, comm_target, sealed_memfd}; use std::collections::HashMap; - use std::io::Read as _; + use std::io::{Read as _, Write as _}; use std::os::fd::AsRawFd as _; use std::os::unix::fs::PermissionsExt as _; + #[test] + fn comm_target_accepts_only_the_callers_own_thread_names() { + let (caller_tid, group) = (4242, 4200); + for (path, expected) in [ + ("/proc/thread-self/comm", Some(caller_tid)), + ("/proc/self/comm", Some(group)), + ("/proc/4200/comm", Some(group)), + ("/proc/self/task/4243/comm", Some(4243)), + ("/proc/4200/task/4243/comm", Some(4243)), + // Another process's thread, or not a comm file. + ("/proc/1/task/1/comm", None), + ("/proc/9999/comm", None), + ("/proc/self/task/4243/environ", None), + ("/proc/self/task/4243/comm/extra", None), + ("/proc/self/mem", None), + ] { + assert_eq!(comm_target(path, caller_tid, group), expected, "{path}"); + } + } + + #[test] + fn workload_thread_rename_through_proc_is_served() { + // Under the workload filter, a thread rename via its own comm file + // is served by the broker and takes effect. + const CHILD_MARKER: &str = "OPENSHELL_THREAD_COMM_CHILD"; + if std::env::var_os(CHILD_MARKER).is_some() { + let (sender, receiver) = std::sync::mpsc::channel(); + let (done, wait) = std::sync::mpsc::channel::<()>(); + let worker = std::thread::spawn(move || { + sender.send(nix::unistd::gettid().as_raw()).unwrap(); + let _ = wait.recv(); + }); + let tid = receiver.recv().unwrap(); + let path = format!("/proc/self/task/{tid}/comm"); + let mut file = std::fs::OpenOptions::new() + .read(true) + .write(true) + .open(&path) + .expect("open own thread comm"); + file.write_all(b"renamed").expect("rename thread"); + let name = std::fs::read_to_string(&path).unwrap(); + done.send(()).unwrap(); + worker.join().unwrap(); + assert_eq!(name.trim(), "renamed"); + return; + } + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = crate::network_broker::NetworkBroker::start_for_test(listener) + .expect("start network broker"); + let status = launcher + .execute(|| { + std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "provider_files::tests::workload_thread_rename_through_proc_is_served", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + }) + .unwrap() + .expect("run workload child"); + assert!( + status.success(), + "thread rename under the workload filter failed" + ); + } + #[test] fn paths_cannot_escape_the_managed_tree() { let valid = "/run/openshell/providers/acme/client.toml"; From 634da46130b677b4a784cd441e3ebdcd01389b97 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 19:30:39 -0700 Subject: [PATCH 12/18] fix(sandbox): keep the broker socket for connected UDP DNS sockets Sending DNS datagrams from the broker's socket required the broker to keep its copy, but the connect paths to the relay and to loopback still released it. glibc connects the resolver socket and then sends A and AAAA together with sendmmsg, which is mediated, so resolution failed. Release the copy on those connect paths only for TCP. Add a regression test for connect followed by sendmmsg that fails fast rather than hanging. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/network_broker.rs | 87 ++++++++++++++++++- 1 file changed, 85 insertions(+), 2 deletions(-) diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 2e5899b983..38bc632c40 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -848,7 +848,11 @@ fn connect_socket( InetKind::Tcp => SocketState::DnsTcp { relay: destination }, InetKind::DnsUdp => SocketState::DnsUdp { relay: destination }, }); - entry.release_preconnect(); + // UDP DNS sockets keep the broker's copy: the broker sends their + // datagrams so ancillary data in workload memory never reaches them. + if kind == InetKind::Tcp { + entry.release_preconnect(); + } return listener.respond_value(notification.id, 0); } // The metadata service lives in the supervisor, even though SDKs address @@ -861,7 +865,10 @@ fn connect_socket( listener.validate_id(notification.id)?; connect_exact(entry.retained_preconnect()?.as_raw_fd(), destination)?; entry.set_state(SocketState::Local { peer: destination }); - entry.release_preconnect(); + // Loopback UDP sends are also broker-sent; keep the copy for them. + if kind == InetKind::Tcp { + entry.release_preconnect(); + } return listener.respond_value(notification.id, 0); } if kind != InetKind::Tcp { @@ -2717,6 +2724,82 @@ mod tests { ); } + #[test] + fn udp_dns_after_connect_sends_with_sendmmsg() { + // glibc connects the resolver socket to the nameserver, then sends A + // and AAAA together with sendmmsg and no destination. sendmmsg is + // mediated, so the broker must still hold its copy of the socket. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let dns_address = broker.dns_address(); + let client = std::thread::spawn(move || { + launcher + .execute(move || -> io::Result>> { + let socket = UdpSocket::bind("0.0.0.0:0")?; + socket.set_read_timeout(Some(Duration::from_secs(5)))?; + socket.connect(dns_address)?; + let queries = [&b"dns-query-a"[..], &b"dns-query-aaaa"[..]]; + let iovecs = queries.map(|query| [io::IoSlice::new(query)]); + let mut controls = [ + rustix::net::SendAncillaryBuffer::default(), + rustix::net::SendAncillaryBuffer::default(), + ]; + let [first, second] = &mut controls; + let mut messages = [ + rustix::net::MMsgHdr::new(&iovecs[0], first), + rustix::net::MMsgHdr::new(&iovecs[1], second), + ]; + let sent = rustix::net::sendmmsg( + &socket, + &mut messages, + rustix::net::SendFlags::empty(), + )?; + if sent != 2 { + return Err(io::Error::other("sendmmsg sent too few")); + } + let mut responses = Vec::new(); + for _ in 0..2 { + let mut response = [0_u8; 32]; + let length = socket.recv(&mut response)?; + responses.push(response[..length].to_vec()); + } + responses.sort(); + Ok(responses) + }) + .expect("launcher result") + }); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("test runtime"); + for _ in 0..2 { + // Bound the wait so a broken send path fails instead of hanging. + let Ok(query) = runtime.block_on(async { + tokio::time::timeout(Duration::from_secs(5), broker.accept_dns()).await + }) else { + let error = client + .join() + .expect("join client") + .expect_err("client must fail when no query arrives"); + panic!("DNS query never reached the relay: {error}"); + }; + let query = query.expect("DNS query"); + let response = if query.request == b"dns-query-a" { + b"dns-response-a".to_vec() + } else if query.request == b"dns-query-aaaa" { + b"dns-response-aaaa".to_vec() + } else { + panic!("unexpected DNS query: {:?}", query.request); + }; + query.complete(Ok(response)).unwrap(); + } + assert_eq!( + client.join().expect("join client").expect("DNS client"), + vec![b"dns-response-a".to_vec(), b"dns-response-aaaa".to_vec()] + ); + } + #[test] fn dns_send_with_ancillary_data_is_refused() { // Ancillary control data (e.g. IP_PKTINFO) can carry a per-message From 94b0296218c1e4e63501365a4fc863a7ee6779aa Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 19:37:42 -0700 Subject: [PATCH 13/18] fix(sandbox): keep slow loopback connects from stalling mediation The loopback connect path held the registry lock while polling a TCP connect for up to five seconds on the single notification dispatcher. A workload connecting to a busy local listener stalled every other mediated syscall, including opens and signals. Connect a duplicate of the retained socket without holding the lock. A nonblocking socket gets the native EINPROGRESS and the kernel completes the handshake on the shared socket; a blocking socket waits on a bounded worker thread. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/network_broker.rs | 160 +++++++++++++++++- 1 file changed, 156 insertions(+), 4 deletions(-) diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 38bc632c40..15b83f4bcb 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -860,15 +860,24 @@ fn connect_socket( if destination.ip().is_loopback() && !openshell_core::google_cloud::is_metadata_destination(destination) { + if kind == InetKind::Tcp { + return connect_local_tcp( + ®istry, + &listener, + notification, + fd, + socket_identity, + destination, + &active_opens, + ); + } + // A UDP connect completes immediately. let mut registry = lock(®istry); let entry = registry.resolve_mut(notification.tid, fd)?; listener.validate_id(notification.id)?; connect_exact(entry.retained_preconnect()?.as_raw_fd(), destination)?; entry.set_state(SocketState::Local { peer: destination }); - // Loopback UDP sends are also broker-sent; keep the copy for them. - if kind == InetKind::Tcp { - entry.release_preconnect(); - } + // Loopback UDP sends are broker-sent; keep the copy for them. return listener.respond_value(notification.id, 0); } if kind != InetKind::Tcp { @@ -950,6 +959,102 @@ fn connect_socket( Ok(()) } +/// Connect a workload TCP socket to a loopback endpoint without blocking the +/// notification dispatcher. +/// +/// The broker connects a duplicate of its retained socket, which shares the +/// workload's open file, and never holds the registry lock while waiting. A +/// nonblocking socket gets the native `EINPROGRESS` and the kernel completes +/// the handshake on the shared socket; a blocking socket waits on a bounded +/// worker thread. A slow or full local listener therefore cannot stall +/// mediation of unrelated syscalls. +fn connect_local_tcp( + registry: &Arc>, + listener: &Arc, + notification: Notification, + fd: RawFd, + socket_identity: SocketIdentity, + destination: SocketAddr, + active_opens: &Arc, +) -> io::Result<()> { + let connector = { + let registry = lock(registry); + let entry = registry.resolve(notification.tid, fd)?; + rustix::io::fcntl_dupfd_cloexec(entry.retained_preconnect()?, 3)? + }; + listener.validate_id(notification.id)?; + // SAFETY: F_GETFL reads the flags of the live shared open file. + let flags = unsafe { libc::fcntl(connector.as_raw_fd(), libc::F_GETFL) }; + if flags < 0 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK != 0 { + let started = with_sockaddr(destination, |pointer, length| { + // SAFETY: pointer/length describe a live sockaddr; the connector + // is a live duplicate of the workload's socket. + if unsafe { libc::connect(connector.as_raw_fd(), pointer, length) } == 0 { + Ok(()) + } else { + Err(io::Error::last_os_error()) + } + }); + let in_progress = started + .as_ref() + .is_err_and(|error| error.raw_os_error() == Some(libc::EINPROGRESS)); + if started.is_ok() || in_progress { + commit_local_connect(registry, notification.tid, fd, socket_identity, destination); + } + return match started { + Ok(()) => listener.respond_value(notification.id, 0), + Err(error) => Err(error), + }; + } + let slot = acquire_pending_open_slot(active_opens)?; + let registry = Arc::clone(registry); + let worker_listener = Arc::clone(listener); + std::thread::Builder::new() + .name("openshell-local-connect".to_string()) + .spawn(move || { + let _slot = slot; + let result = connect_exact(connector.as_raw_fd(), destination); + if result.is_ok() { + commit_local_connect( + ®istry, + notification.tid, + fd, + socket_identity, + destination, + ); + } + let _ = match result { + Ok(()) => worker_listener.respond_value(notification.id, 0), + Err(error) => { + worker_listener.respond_errno(notification.id, error_to_errno(&error)) + } + }; + }) + .map_err(|error| io::Error::other(format!("start local-connect worker: {error}")))?; + Ok(()) +} + +/// Record a completed or in-progress loopback TCP connect, unless the +/// descriptor now names a different socket. +fn commit_local_connect( + registry: &Mutex, + tid: u32, + fd: RawFd, + socket_identity: SocketIdentity, + destination: SocketAddr, +) { + let mut registry = lock(registry); + if let Ok(entry) = registry.resolve_mut(tid, fd) + && entry.identity() == socket_identity + { + entry.set_state(SocketState::Local { peer: destination }); + entry.release_preconnect(); + } +} + /// Result for a `connect` on a socket the broker already connected, as the /// kernel would report it: `Some(0)` for success, `Some(errno)` for an error, /// `None` when the socket is not yet connected. @@ -2567,6 +2672,53 @@ mod tests { assert_eq!(run_workload(&launcher), "kill=0 tgkill=0"); } + #[test] + fn slow_loopback_connect_does_not_stall_other_mediation() { + // A listener that never accepts, with a full backlog, makes further + // connects wait. Other mediated syscalls must not wait behind them, + // whether the pending connect is blocking or nonblocking. + let saturated = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None).unwrap(); + saturated + .bind(&"127.0.0.1:0".parse::().unwrap().into()) + .unwrap(); + saturated.listen(0).unwrap(); + let address = saturated.local_addr().unwrap().as_socket().unwrap(); + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let (blocking_wait, nonblocking_result) = launcher + .execute(move || { + // Fill the accept queue; later connects to it stall. + let filled = TcpStream::connect(address).expect("fill accept queue"); + let pending = std::thread::spawn(move || TcpStream::connect(address)); + std::thread::sleep(Duration::from_millis(500)); + let started = Instant::now(); + drop(UdpSocket::bind("127.0.0.1:0")); + let blocking_wait = started.elapsed(); + // A nonblocking connect reports progress immediately. + let nonblocking = socket2::Socket::new( + socket2::Domain::IPV4, + socket2::Type::STREAM.nonblocking(), + None, + ) + .unwrap(); + let nonblocking_result = nonblocking + .connect(&address.into()) + .err() + .and_then(|error| error.raw_os_error()); + drop(pending.join()); + drop(filled); + (blocking_wait, nonblocking_result) + }) + .expect("launcher result"); + assert!( + blocking_wait < Duration::from_secs(2), + "mediated socket creation waited {blocking_wait:?} behind a slow connect" + ); + assert_eq!(nonblocking_result, Some(libc::EINPROGRESS)); + } + #[test] fn workload_cannot_bind_a_non_loopback_source_address() { // Workload sockets can only present loopback source addresses. A From afbd6e369ae71fbdfd10c2d780a24efdaea9cf9a Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 22:03:07 -0700 Subject: [PATCH 14/18] fix(sandbox): contain network broker handler panics A panic in a notification handler unwound the single broker thread while the health flag still read healthy, so every later blocked workload syscall hung until the sandbox was killed. Run each dispatch under catch_unwind: a panicking handler fails only that syscall with EIO and the broker keeps mediating. If the broker thread ever exits, mark it unhealthy so dependent operations fail closed instead of blocking. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/network_broker.rs | 101 +++++++++++++++--- 1 file changed, 86 insertions(+), 15 deletions(-) diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 15b83f4bcb..5012122a9a 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -40,6 +40,11 @@ const DNS_RELAY_ADDRESS: SocketAddr = SocketAddr::V4(std::net::SocketAddrV4::new const RELAY_CONNECT_TIMEOUT: Duration = Duration::from_secs(5); const NETWORK_DECISION_TIMEOUT: Duration = Duration::from_secs(30); +#[cfg(test)] +const PANIC_SENTINEL_LEVEL: i32 = libc::IPPROTO_TCP; +#[cfg(test)] +const PANIC_SENTINEL_OPTION: i32 = 0x7fff_fffe; + fn retry_notification_receive(error: &io::Error) -> bool { error.kind() == io::ErrorKind::Interrupted || error.raw_os_error() == Some(libc::ENOENT) } @@ -266,23 +271,47 @@ impl NetworkBroker { break; } }; - if let Err(error) = dispatch_notification( - Arc::clone(®istry), - Arc::clone(&listener), - notification, - queues.clone(), - ) { - tracing::warn!( - tid = notification.tid, - syscall = notification.syscall, - %error, - "sandbox network notification denied (tid={}, syscall={}): {error}", - notification.tid, - notification.syscall - ); - let _ = listener.respond_errno(notification.id, error_to_errno(&error)); + // Contain a handler panic so one faulty notification + // cannot silently kill the broker and hang every blocked + // workload syscall. The failing syscall gets an error; the + // broker keeps mediating the rest. + let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + dispatch_notification( + Arc::clone(®istry), + Arc::clone(&listener), + notification, + queues.clone(), + ) + })); + match outcome { + Ok(Ok(())) => {} + Ok(Err(error)) => { + tracing::warn!( + tid = notification.tid, + syscall = notification.syscall, + %error, + "sandbox network notification denied (tid={}, syscall={}): {error}", + notification.tid, + notification.syscall + ); + let _ = + listener.respond_errno(notification.id, error_to_errno(&error)); + } + Err(_) => { + tracing::error!( + tid = notification.tid, + syscall = notification.syscall, + "sandbox network notification handler panicked (tid={}, syscall={})", + notification.tid, + notification.syscall + ); + let _ = listener.respond_errno(notification.id, libc::EIO); + } } } + // The broker thread is exiting; dependent operations must fail + // closed rather than block on a listener no one services. + broker_healthy.store(false, Ordering::Release); }) .map_err(|error| io::Error::other(format!("start network broker: {error}")))?; Ok(Self { @@ -570,6 +599,11 @@ fn dispatch_notification( .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; let option = i32::try_from(notification.args[2]) .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; + #[cfg(test)] + assert!( + !(level == PANIC_SENTINEL_LEVEL && option == PANIC_SENTINEL_OPTION), + "test-only setsockopt panic sentinel" + ); if socket_option_is_denied(level, option) { return Err(io::Error::from_raw_os_error(libc::EPERM)); } @@ -2719,6 +2753,43 @@ mod tests { assert_eq!(nonblocking_result, Some(libc::EINPROGRESS)); } + #[test] + fn a_panicking_handler_does_not_kill_the_broker() { + // A handler panic must fail only that syscall, not hang every later + // mediated syscall by silently killing the broker thread. setsockopt + // is routed through a hook that panics in test builds on a sentinel + // option, standing in for an unexpected handler bug. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let (panicked, after) = launcher + .execute(|| { + let socket = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None) + .unwrap(); + // SAFETY: scalar setsockopt with the test-only panic sentinel. + let value = 1_i32; + let panicked = unsafe { + libc::setsockopt( + socket.as_raw_fd(), + PANIC_SENTINEL_LEVEL, + PANIC_SENTINEL_OPTION, + (&raw const value).cast(), + libc::socklen_t::try_from(size_of::()).unwrap(), + ) + }; + // A later mediated syscall still completes, proving the broker + // survived the panic. + let after = socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::DGRAM, None) + .map(drop) + .map_err(|error| error.raw_os_error()); + (panicked, after) + }) + .expect("launcher result"); + assert_eq!(panicked, -1, "panicking syscall must fail"); + assert_eq!(after, Ok(()), "broker must keep mediating after a panic"); + } + #[test] fn workload_cannot_bind_a_non_loopback_source_address() { // Workload sockets can only present loopback source addresses. A From 7a26c209aac0c6ef19fe4079484fe0e5d01412d3 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 22:07:22 -0700 Subject: [PATCH 15/18] fix(sandbox): reject ancillary data on DNS sends instead of broker-sending Sending mediated DNS datagrams from the broker's own socket required keeping a broker handle for each DNS socket's whole life, which regressed real name resolution through reclaim and resource accounting that the mock-based unit tests did not exercise. Revert to continuing the kernel send, but refuse a send that carries ancillary control data (msg_controllen != 0) at read time, so a per-message routing override such as IP_PKTINFO cannot ride a mediated DNS send. A loopback destination contains an override that races the check. Full broker-side UDP mediation is left to a separate change with deployment e2e. Signed-off-by: Drew Newberry --- .../openshell-sandbox/src/network_broker.rs | 139 ++++-------------- 1 file changed, 25 insertions(+), 114 deletions(-) diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 5012122a9a..0a64fd99d7 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -882,11 +882,7 @@ fn connect_socket( InetKind::Tcp => SocketState::DnsTcp { relay: destination }, InetKind::DnsUdp => SocketState::DnsUdp { relay: destination }, }); - // UDP DNS sockets keep the broker's copy: the broker sends their - // datagrams so ancillary data in workload memory never reaches them. - if kind == InetKind::Tcp { - entry.release_preconnect(); - } + entry.release_preconnect(); return listener.respond_value(notification.id, 0); } // The metadata service lives in the supervisor, even though SDKs address @@ -911,7 +907,7 @@ fn connect_socket( listener.validate_id(notification.id)?; connect_exact(entry.retained_preconnect()?.as_raw_fd(), destination)?; entry.set_state(SocketState::Local { peer: destination }); - // Loopback UDP sends are broker-sent; keep the copy for them. + entry.release_preconnect(); return listener.respond_value(notification.id, 0); } if kind != InetKind::Tcp { @@ -1316,8 +1312,6 @@ fn classify_send( libc::SYS_sendmsg => vec![read_sendmsg_message( notification.tid, notification.args[1], - i32::try_from(notification.args[2]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, )?], libc::SYS_sendmmsg => read_sendmmsg_messages(notification)?, _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), @@ -1330,15 +1324,11 @@ fn classify_send( if entry.metadata().kind == InetKind::DnsUdp && matches!(entry.state(), SocketState::Local { .. }) => { - if !messages.iter().all(|message| message.destination.is_none()) { - return Err(io::Error::from_raw_os_error(libc::EACCES)); - } - let source_fd = entry.retained_preconnect()?.as_raw_fd(); - listener.validate_id(notification.id)?; - for message in &messages { - send_dns_message(source_fd, message)?; + if messages.iter().all(|message| message.destination.is_none()) { + listener.respond_continue(notification.id) + } else { + Err(io::Error::from_raw_os_error(libc::EACCES)) } - listener.respond_value(notification.id, dns_send_result(syscall, &messages)) } Ok(entry) if matches!(entry.state(), SocketState::DnsUdp { .. }) => { let SocketState::DnsUdp { relay } = entry.state() else { @@ -1351,20 +1341,15 @@ fn classify_send( // destination is absent or names that same relay. The mandatory // outer network fence remains the fail-closed backstop for the // sibling-thread pointer race inherent in seccomp CONTINUE. - let relay = *relay; - if !messages.iter().all(|message| { + if messages.iter().all(|message| { message .destination - .is_none_or(|destination| destination == relay) + .is_none_or(|destination| destination == *relay) }) { - return Err(io::Error::from_raw_os_error(libc::EACCES)); - } - let source_fd = entry.retained_preconnect()?.as_raw_fd(); - listener.validate_id(notification.id)?; - for message in &messages { - send_dns_message(source_fd, message)?; + listener.respond_continue(notification.id) + } else { + Err(io::Error::from_raw_os_error(libc::EACCES)) } - listener.respond_value(notification.id, dns_send_result(syscall, &messages)) } Ok(entry) if entry.metadata().kind == InetKind::DnsUdp @@ -1387,15 +1372,15 @@ fn classify_send( lock(&dns_relay.udp_admissions).remove(&peer); return Err(error); } - for message in &messages { - send_dns_message(source_fd, message)?; - } - // Keep the broker's retained socket so later sends also originate - // from it; the workload's aliased fd shares this open file. + // The socket is pinned to the relay and bound to loopback. The + // kernel performs the send; ancillary data was rejected at read + // time and a loopback destination contains any per-message + // routing override that races the check. entry.set_state(SocketState::DnsUdp { relay: dns_relay.address, }); - listener.respond_value(notification.id, dns_send_result(syscall, &messages)) + entry.release_preconnect(); + listener.respond_continue(notification.id) } Ok(_) => Err(io::Error::from_raw_os_error(libc::EDESTADDRREQ)), // Non-INET sockets and natively accepted sockets were never @@ -1422,19 +1407,10 @@ fn send_flags(syscall: i64, args: [u64; 6]) -> i32 { } struct SendMessage { - data: Vec, destination: Option, - flags: i32, } fn read_sendto_message(notification: Notification) -> io::Result { - let length = usize::try_from(notification.args[2]) - .map_err(|_| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - if u16::try_from(length).is_err() { - return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); - } - let mut data = vec![0_u8; length]; - task_memory::read_exact(notification.tid, notification.args[1], &mut data)?; let destination = if notification.args[4] == 0 { None } else { @@ -1444,19 +1420,15 @@ fn read_sendto_message(notification: Notification) -> io::Result { notification.args[5], )?) }; - Ok(SendMessage { - data, - destination, - flags: i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, - }) + Ok(SendMessage { destination }) } -fn read_sendmsg_message(tid: u32, address: u64, flags: i32) -> io::Result { +fn read_sendmsg_message(tid: u32, address: u64) -> io::Result { let header = read_task_value::(tid, address)?; - // Ancillary data can carry per-message routing overrides (IP_PKTINFO). - // The broker sends from its own socket without control messages, so a - // workload that needs them is refused rather than silently stripped. + // Ancillary data can carry a per-message routing override (IP_PKTINFO). + // Refuse it rather than continue a send the broker did not inspect; a + // loopback destination additionally contains an override that races this + // check. if header.msg_controllen != 0 { return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); } @@ -1469,38 +1441,7 @@ fn read_sendmsg_message(tid: u32, address: u64, flags: i32) -> io::Result 32 { - return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); - } - let mut data = Vec::new(); - for index in 0..iov_count { - let offset = index - .checked_mul(size_of::()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?; - let iov = read_task_value::( - tid, - (header.msg_iov as u64) - .checked_add(u64::try_from(offset).unwrap_or(u64::MAX)) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?, - )?; - let start = data.len(); - let end = start - .checked_add(iov.iov_len) - .filter(|length| u16::try_from(*length).is_ok()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - data.resize(end, 0); - task_memory::read_exact(tid, iov.iov_base as u64, &mut data[start..end])?; - } - Ok(SendMessage { - data, - destination, - flags, - }) + Ok(SendMessage { destination }) } fn read_sendmmsg_messages(notification: Notification) -> io::Result> { @@ -1509,8 +1450,6 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result 32 { return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); } - let flags = i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; (0..count) .map(|index| { let offset = index @@ -1519,39 +1458,11 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result io::Result<()> { - // Strip MSG_FASTOPEN (already rejected) and MSG_MORE (no corking here). - let flags = message.flags & !(libc::MSG_FASTOPEN | libc::MSG_MORE); - // SAFETY: `fd` is the retained relay-connected UDP socket; the buffer is - // live for the duration of the call. - let sent = unsafe { libc::send(fd, message.data.as_ptr().cast(), message.data.len(), flags) }; - if sent < 0 { - return Err(io::Error::last_os_error()); - } - if usize::try_from(sent).ok() == Some(message.data.len()) { - Ok(()) - } else { - Err(io::Error::from_raw_os_error(libc::EIO)) - } -} - -fn dns_send_result(syscall: i64, messages: &[SendMessage]) -> i64 { - if syscall == libc::SYS_sendmmsg { - i64::try_from(messages.len()).unwrap_or(i64::MAX) - } else { - i64::try_from(messages.first().map_or(0, |message| message.data.len())).unwrap_or(i64::MAX) - } -} - fn read_task_value(tid: u32, address: u64) -> io::Result { let mut bytes = vec![0_u8; size_of::()]; task_memory::read_exact(tid, address, &mut bytes)?; From d7d449908f9bd03982efde21c0a07a3ce77892e8 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 22:33:44 -0700 Subject: [PATCH 16/18] fix(sandbox): address review findings across the hardening changes Fixes: - The blocking local-connect worker no longer toggles O_NONBLOCK on the open file it shares with the workload; it waits with a plain connect. - tgkill and rt_tgsigqueueinfo are notified only when they send SIGCONT, so ordinary thread signals such as Go preemption stay in the kernel. The frozen flag is re-read immediately before delivery. - A shell redirect to the caller's own thread comm file works again; O_CREAT and O_TRUNC are no-ops there and O_EXCL returns EEXIST. - The teardown scan is a linear walk and fails closed when /proc cannot be read. Simplifications: - Remove tests that need infrastructure outside the repo or prove nothing: the topology harness tests, the static-server benchmark, the direct-syscall accept test, the comm rename test without Landlock, a serde-default test, and the test-only panic hook in setsockopt. - Drop the unused Failed connect outcome, the redundant read-back after binding to loopback, redundant OPENSHELL_SANDBOX settings, and stale or duplicated comments; reuse the reserved environment prefix constant. - Tighten the support matrix and OpenShift wording. Signed-off-by: Drew Newberry --- .../src/linux/child_seccomp.rs | 5 +- .../src/linux/process_signal.rs | 79 ++++--- .../src/linux/seccomp_notify.rs | 32 ++- .../src/linux/socket_confinement.rs | 219 ++--------------- .../src/boundary_protocol.rs | 10 - crates/openshell-sandbox/src/boundary_exec.rs | 4 +- crates/openshell-sandbox/src/boundary_io.rs | 72 +++--- .../openshell-sandbox/src/boundary_server.rs | 19 +- .../openshell-sandbox/src/network_broker.rs | 223 +++--------------- crates/openshell-sandbox/src/process.rs | 12 +- .../openshell-sandbox/src/provider_files.rs | 67 +----- docs/about/support-matrix.mdx | 2 +- docs/kubernetes/openshift.mdx | 3 +- 13 files changed, 177 insertions(+), 570 deletions(-) diff --git a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs index e6ce09d9f6..6fc9f67f73 100644 --- a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs +++ b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs @@ -124,8 +124,9 @@ pub fn mark_inherited_descriptors_close_on_exec() -> io::Result<()> { /// `sandbox_tgid` is the sandbox PID as visible from its workload namespace. /// The filter blocks thread-targeting operations that name the trusted sandbox /// leader and blocks process-directed operations with the same target. The -/// ordinary workload listener additionally mediates `kill`, `tkill`, and -/// `rt_sigqueueinfo`: Linux accepts nonleader TIDs for these operations, so a +/// ordinary workload listener additionally mediates `kill`, `tkill`, +/// `rt_sigqueueinfo`, and `SIGCONT` sent with `tgkill` or +/// `rt_tgsigqueueinfo`: Linux accepts nonleader TIDs for these operations, so a /// static TGID comparison alone cannot protect future sandbox worker threads. pub fn prepare(sandbox_tgid: u32) -> io::Result { if sandbox_tgid == 0 { diff --git a/crates/openshell-isolation-interface/src/linux/process_signal.rs b/crates/openshell-isolation-interface/src/linux/process_signal.rs index 67fcae68b1..f2addd54dc 100644 --- a/crates/openshell-isolation-interface/src/linux/process_signal.rs +++ b/crates/openshell-isolation-interface/src/linux/process_signal.rs @@ -12,6 +12,7 @@ use std::io; use std::os::fd::{AsRawFd, FromRawFd, OwnedFd}; +use std::sync::atomic::{AtomicBool, Ordering}; use crate::linux::seccomp_notify::{Notification, NotificationListener}; use crate::linux::task_memory; @@ -27,7 +28,7 @@ pub fn mediate_process_signal( listener: &NotificationListener, notification: Notification, sandbox_tgid: u32, - workload_frozen: bool, + workload_frozen: &AtomicBool, ) -> io::Result<()> { listener.validate_id(notification.id)?; let target = scalar_int(notification.args[0]); @@ -66,6 +67,7 @@ pub fn mediate_process_signal( _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), }; listener.validate_id(notification.id)?; + refuse_resume_while_frozen(signal, workload_frozen)?; // SAFETY: retained owns a live pidfd; info is null or a complete trusted // copy. The kernel targets that process object, never a reused numeric PID. let result = unsafe { @@ -83,59 +85,57 @@ pub fn mediate_process_signal( listener.respond_value(notification.id, 0) } -/// Continue a positive-target `tkill` only when the target thread belongs to -/// an untrusted workload process rather than the sandbox runtime itself. +/// Continue a thread-directed signal (`tkill`, `tgkill`, +/// `rt_tgsigqueueinfo`) aimed at an untrusted workload thread. /// /// Continuing preserves Linux's thread-directed signal semantics, including -/// the cancellation signal used by musl. A target that exits between the -/// ownership check and continuation can only be reused inside the same PID -/// namespace; the static child filter still rejects the sandbox leader. +/// the cancellation signal used by musl. The static child filter rejects the +/// sandbox leader as a `tgkill`/`rt_tgsigqueueinfo` group, and the kernel +/// rejects a thread outside the named group, so those two are notified only +/// for `SIGCONT`. A `tkill` names a bare thread, so its group is resolved here; +/// a target reused between this check and continuation stays inside the same +/// PID namespace. pub fn mediate_thread_signal( listener: &NotificationListener, notification: Notification, sandbox_tgid: u32, - workload_frozen: bool, + workload_frozen: &AtomicBool, ) -> io::Result<()> { listener.validate_id(notification.id)?; - // tkill(tid, sig); tgkill(tgid, tid, sig); rt_tgsigqueueinfo(tgid, tid, - // sig, info). The kernel itself validates a queued siginfo's code. - let (claimed_group, target, signal) = match i64::from(notification.syscall) { - libc::SYS_tkill => ( - None, - scalar_int(notification.args[0]), - scalar_int(notification.args[1]), - ), - libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo => ( - Some(scalar_int(notification.args[0])), - scalar_int(notification.args[1]), - scalar_int(notification.args[2]), - ), + let signal = match i64::from(notification.syscall) { + libc::SYS_tkill => { + let target = scalar_int(notification.args[0]); + let signal = scalar_int(notification.args[1]); + if target <= 0 { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } + let target = + u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; + let target_group = thread_group_id(target)?; + if target_group == sandbox_tgid || target_group == 0 { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } + signal + } + libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo => scalar_int(notification.args[2]), _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), }; - if target <= 0 || claimed_group.is_some_and(|group| group <= 0) { - return Err(io::Error::from_raw_os_error(libc::EPERM)); - } if !(0..=64).contains(&signal) { return Err(io::Error::from_raw_os_error(libc::EINVAL)); } - refuse_resume_while_frozen(signal, workload_frozen)?; - let target = u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; - let target_group = thread_group_id(target)?; - if target_group == sandbox_tgid || target_group == 0 { - return Err(io::Error::from_raw_os_error(libc::EPERM)); - } - if claimed_group.is_some_and(|group| u32::try_from(group).ok() != Some(target_group)) { - // The kernel reports a thread outside the named group as missing. - return Err(io::Error::from_raw_os_error(libc::ESRCH)); - } listener.validate_id(notification.id)?; + refuse_resume_while_frozen(signal, workload_frozen)?; listener.respond_continue(notification.id) } /// While the boundary has stopped the workload for supervisor recovery, a /// workload process that was not yet stopped must not resume the others. -fn refuse_resume_while_frozen(signal: i32, workload_frozen: bool) -> io::Result<()> { - if workload_frozen && signal == libc::SIGCONT { +/// +/// The flag is read immediately before delivery. The freezer does not wait +/// for in-flight notifications, so a signal already past this check when the +/// freeze begins can still be delivered. +fn refuse_resume_while_frozen(signal: i32, workload_frozen: &AtomicBool) -> io::Result<()> { + if signal == libc::SIGCONT && workload_frozen.load(Ordering::Acquire) { return Err(io::Error::from_raw_os_error(libc::EPERM)); } Ok(()) @@ -205,8 +205,13 @@ mod tests { .recv_timeout(std::time::Duration::from_secs(5)) .unwrap(); let notification = listener.receive().unwrap(); - let error = - mediate_process_signal(&listener, notification, std::process::id(), false).unwrap_err(); + let error = mediate_process_signal( + &listener, + notification, + std::process::id(), + &AtomicBool::new(false), + ) + .unwrap_err(); assert_eq!(error.raw_os_error(), Some(libc::EPERM)); listener .respond_errno(notification.id, libc::EPERM) diff --git a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs index 0e447bc817..5ae24cd27d 100644 --- a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs +++ b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs @@ -170,9 +170,10 @@ impl NotificationProbeReport { /// Owned listener returned by `SECCOMP_FILTER_FLAG_NEW_LISTENER`. /// /// The listener is installed without `WAIT_KILLABLE_RECV`, so mediation is the -/// same on every kernel. A signal can interrupt a notified syscall and the -/// kernel then restarts it; broker handlers check the notification is still -/// live before acting and answer a repeated operation as the kernel would. +/// same on every kernel. A signal can interrupt a notified syscall, which the +/// kernel then restarts or fails with `EINTR`; broker handlers check the +/// notification is still live before acting and answer a repeated operation +/// as the kernel would. pub struct NotificationListener { fd: OwnedFd, } @@ -705,6 +706,10 @@ fn build_filter(syscalls: &[i64]) -> io::Result> { append_sendto_filter(&mut program)?; continue; } + if matches!(syscall, libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo) { + append_resume_signal_filter(&mut program, syscall)?; + continue; + } let syscall = u32::try_from(syscall) .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "negative syscall number"))?; program.extend([ @@ -745,6 +750,27 @@ fn append_sendto_filter(program: &mut Vec) -> io::Result<()> Ok(()) } +/// Notify a thread-group signal only when it sends `SIGCONT`, which the broker +/// refuses while the workload is frozen. Every other signal (for example Go's +/// preemption signal) stays in the kernel. +fn append_resume_signal_filter( + program: &mut Vec, + syscall: i64, +) -> io::Result<()> { + let syscall = u32::try_from(syscall) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "negative syscall number"))?; + let resume = u32::try_from(libc::SIGCONT) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "negative signal number"))?; + program.extend([ + jump(BPF_JMP_JEQ_K, syscall, 0, 4), + stmt(BPF_LD_W_ABS, argument_word_offset(2, 0)), + jump(BPF_JMP_JEQ_K, resume, 0, 1), + stmt(BPF_RET_K, SECCOMP_RET_USER_NOTIF), + stmt(BPF_RET_K, SECCOMP_RET_ALLOW), + ]); + Ok(()) +} + const fn argument_word_offset(argument: u32, word: u32) -> u32 { SECCOMP_DATA_ARGS_OFFSET + argument * 8 + word * 4 } diff --git a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs index 294eec3939..715ce4ca9b 100644 --- a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs +++ b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs @@ -19,20 +19,14 @@ use socket2::{Domain, SockFilter, SockRef, Socket, Type}; const LOOPBACK_DEVICE: &[u8] = b"lo"; -/// Bind `fd` to the loopback device and verify the kernel recorded it. +/// Bind `fd` to the loopback device. /// /// # Errors /// -/// Returns the kernel error when the binding cannot be installed, or `EPERM` -/// when the socket is already bound to another device. +/// Returns the kernel error, including `EPERM` when the socket is already +/// bound to a device. pub fn confine_to_loopback(fd: impl AsFd) -> io::Result<()> { - let socket = SockRef::from(&fd); - socket.bind_device(Some(LOOPBACK_DEVICE))?; - if socket.device()?.as_deref() == Some(LOOPBACK_DEVICE) { - Ok(()) - } else { - Err(io::Error::from_raw_os_error(libc::EPERM)) - } + SockRef::from(&fd).bind_device(Some(LOOPBACK_DEVICE)) } /// Return the device name `fd` is bound to, or `None` when unbound. @@ -94,9 +88,7 @@ const fn filter(code: u32, jt: u8, jf: u8, k: u32) -> SockFilter { /// proves that the sandbox credentials cannot clear or replace it. For IPv4 /// and IPv6 it proves that a stream accepted from a confined listener inherits /// the binding and keeps it after an `AF_UNSPEC` disconnect. IPv6 is skipped -/// only when the kernel or namespace does not provide it. Routed ingress and -/// egress cannot be exercised without a non-loopback route, so those -/// guarantees rest on the kernel behavior verified per target kernel. +/// only when the kernel or namespace does not provide it. /// /// # Errors /// @@ -158,11 +150,8 @@ fn probe_accept_inherits_binding() -> io::Result<()> { }; confine_to_loopback(&listener) .map_err(|error| probe_error("confine probe listener", &error))?; - let client = std::net::TcpStream::connect(listener.local_addr()?)?; - let (accepted, peer) = listener.accept()?; - if peer != client.local_addr()? { - return Err(io::Error::other("accepted probe peer mismatch")); - } + let _client = std::net::TcpStream::connect(listener.local_addr()?)?; + let (accepted, _) = listener.accept()?; if bound_device(&accepted)?.as_deref() != Some(LOOPBACK_DEVICE) { return Err(io::Error::other( "accepted socket did not inherit the loopback binding", @@ -199,6 +188,10 @@ mod tests { #[test] fn active_probe_passes_without_capabilities() { + // Root holds CAP_NET_RAW, which may change a device binding. + if rustix::process::geteuid().is_root() { + return; + } probe_loopback_confinement().expect("loopback confinement probe"); } @@ -210,6 +203,9 @@ mod tests { #[test] fn confined_socket_cannot_be_rebound() { + if rustix::process::geteuid().is_root() { + return; + } let socket = new_socket(Domain::IPV4, Type::DGRAM); confine_to_loopback(&socket).unwrap(); assert!(confine_to_loopback(&socket).is_err()); @@ -260,13 +256,8 @@ mod tests { #[test] fn confined_listener_peers_are_limited_to_this_network_namespace() { - // Packets that arrive on another interface never match a listener - // bound to loopback. The only non-loopback peer address it can see is - // a client in the same namespace that binds its source to a local - // non-loopback address and connects to loopback; that client could - // equally connect from 127.0.0.1. Workload sockets cannot bind such a - // source, because the broker only permits loopback or unspecified - // binds. + // Only a same-namespace client can reach a loopback-bound listener; + // connecting to the host's own address is refused. let Some(local) = local_non_loopback_address() else { eprintln!("skipping: no non-loopback IPv4 address in this namespace"); return; @@ -299,182 +290,4 @@ mod tests { assert_eq!(peer, client.local_addr().unwrap().as_socket().unwrap()); assert_eq!(peer.ip(), std::net::IpAddr::V4(local)); } - - /// Opt-in checks against a real non-loopback topology. - /// - /// A trusted harness creates a network namespace whose non-loopback - /// device routes to a peer namespace, enables forwarding and the other - /// permissive routing sysctls, and runs these tests unprivileged inside - /// it. The harness, not the workload, holds any privileges. Each check - /// carries an unconfined positive control so a broken observer cannot - /// pass vacuously. For egress, the harness captures UDP/TCP port 9 on - /// the peer side: exactly one datagram, the unconfined control payload, - /// must arrive. - mod topology { - use super::*; - use std::net::{IpAddr, UdpSocket}; - use std::time::Instant; - - fn env(name: &str) -> String { - std::env::var(name).unwrap_or_else(|_| panic!("{name} is required")) - } - - fn tx_packets(device: &str) -> u64 { - let table = std::fs::read_to_string("/proc/net/dev").unwrap(); - let line = table - .lines() - .find(|line| line.trim_start().starts_with(&format!("{device}:"))) - .unwrap_or_else(|| panic!("{device} is not in this network namespace")); - // Receive has eight columns; transmit packets is the tenth field. - line.split(':') - .nth(1) - .unwrap() - .split_whitespace() - .nth(9) - .unwrap() - .parse() - .unwrap() - } - - fn quiet_counter(device: &str) -> u64 { - let deadline = Instant::now() + Duration::from_secs(5); - loop { - let before = tx_packets(device); - std::thread::sleep(Duration::from_millis(300)); - let after = tx_packets(device); - if before == after { - return after; - } - assert!(Instant::now() < deadline, "{device} never became quiet"); - } - } - - fn confined(domain: Domain, kind: Type) -> Socket { - let socket = new_socket(domain, kind.nonblocking()); - confine_to_loopback(&socket).unwrap(); - socket - } - - fn errno(result: io::Result) -> Option { - result.err().map(|error| error.raw_os_error().unwrap_or(0)) - } - - fn attempt_egress(socket: &Socket, kind: Type, destination: SocketAddr) -> Option { - // Streams use Fast Open so the send itself attempts a connect. - let flags = if kind == Type::DGRAM { - 0 - } else { - libc::MSG_FASTOPEN - }; - errno(socket.send_to_with_flags(b"probe", &destination.into(), flags)) - } - - #[test] - #[ignore = "requires the privileged topology harness"] - fn topology_confined_sockets_emit_nothing_on_routed_devices() { - let device = env("OPENSHELL_TOPOLOGY_DEVICE"); - let destinations: Vec = env("OPENSHELL_TOPOLOGY_DESTINATIONS") - .split(',') - .map(|value| value.parse().unwrap()) - .collect(); - - // Positive control: an unconfined datagram to the first - // destination is observed on the routed device. - let baseline = quiet_counter(&device); - let unconfined = UdpSocket::bind(match destinations[0].ip() { - IpAddr::V4(_) => "0.0.0.0:0", - IpAddr::V6(_) => "[::]:0", - }) - .unwrap(); - unconfined.send_to(b"control", destinations[0]).unwrap(); - let deadline = Instant::now() + Duration::from_secs(2); - while tx_packets(&device) == baseline { - assert!( - Instant::now() < deadline, - "observer missed the unconfined control" - ); - std::thread::sleep(Duration::from_millis(20)); - } - - let baseline = quiet_counter(&device); - let mut outcomes = Vec::new(); - for destination in &destinations { - let domain = Domain::for_address(*destination); - for kind in [Type::DGRAM, Type::STREAM] { - let socket = confined(domain, kind); - outcomes.push(( - destination, - kind, - "send", - attempt_egress(&socket, kind, *destination), - )); - if kind == Type::STREAM { - let socket = confined(domain, kind); - let result = socket.connect(&(*destination).into()); - outcomes.push((destination, kind, "connect", errno(result))); - std::thread::sleep(Duration::from_millis(200)); - } - } - } - std::thread::sleep(Duration::from_millis(500)); - let after = tx_packets(&device); - for outcome in &outcomes { - eprintln!("confined attempt {outcome:?}"); - } - // Link-level chatter (MLD, neighbor discovery) also moves this - // counter, so it cannot prove a negative on its own. The harness - // captures the probe port on the peer side as the authoritative - // observer; report the delta for correlation. - eprintln!("confined phase {device} tx delta {}", after - baseline); - } - - #[test] - #[ignore = "requires the privileged topology harness"] - fn topology_confined_listeners_reject_routed_ingress() { - let confined_port: u16 = env("OPENSHELL_TOPOLOGY_CONFINED_PORT").parse().unwrap(); - let control_port: u16 = env("OPENSHELL_TOPOLOGY_CONTROL_PORT").parse().unwrap(); - let wait = Duration::from_secs(env("OPENSHELL_TOPOLOGY_WAIT_SECS").parse().unwrap()); - let listen = |port: u16, confine: bool| { - let socket = new_socket(Domain::IPV6, Type::STREAM); - // Dual-stack wildcard covers IPv4 and IPv4-mapped peers too. - socket.set_only_v6(false).unwrap(); - if confine { - confine_to_loopback(&socket).unwrap(); - } - socket - .bind(&SocketAddr::from((std::net::Ipv6Addr::UNSPECIFIED, port)).into()) - .unwrap(); - socket.listen(16).unwrap(); - socket.set_nonblocking(true).unwrap(); - TcpListener::from(socket) - }; - let confined_listener = listen(confined_port, true); - let control_listener = listen(control_port, false); - let deadline = Instant::now() + wait; - let (mut confined_peers, mut control_peers) = (Vec::new(), Vec::new()); - while Instant::now() < deadline { - while let Ok((_, peer)) = confined_listener.accept() { - confined_peers.push(peer); - } - while let Ok((_, peer)) = control_listener.accept() { - control_peers.push(peer); - } - std::thread::sleep(Duration::from_millis(20)); - } - eprintln!("control accepted {control_peers:?}; confined accepted {confined_peers:?}"); - assert!( - control_peers.iter().any(|peer| !peer.ip().is_loopback() - && peer.ip() != IpAddr::from(std::net::Ipv6Addr::LOCALHOST)), - "positive control saw no routed client" - ); - assert!( - confined_peers.iter().all(|peer| match peer.ip() { - IpAddr::V4(ip) => ip.is_loopback(), - IpAddr::V6(ip) => - ip.is_loopback() || ip.to_ipv4_mapped().is_some_and(|ip| ip.is_loopback()), - }), - "confined listener accepted a routed client" - ); - } - } } diff --git a/crates/openshell-sandbox-backend/src/boundary_protocol.rs b/crates/openshell-sandbox-backend/src/boundary_protocol.rs index 369fe973d5..0cf4efcec8 100644 --- a/crates/openshell-sandbox-backend/src/boundary_protocol.rs +++ b/crates/openshell-sandbox-backend/src/boundary_protocol.rs @@ -1494,16 +1494,6 @@ mod tests { assert!(!audit.properties().egress_interception.enforced); } - #[test] - fn audit_evidence_rejects_missing_socket_loopback_confinement_field() { - let mut value = serde_json::to_value(complete_audit_evidence()).unwrap(); - value - .as_object_mut() - .unwrap() - .remove("socket_loopback_confinement"); - assert!(serde_json::from_value::(value).is_err()); - } - #[test] fn binary_identity_wire_rejects_ambiguous_or_invalid_shapes() { for encoded in [ diff --git a/crates/openshell-sandbox/src/boundary_exec.rs b/crates/openshell-sandbox/src/boundary_exec.rs index da35294c06..5482629069 100644 --- a/crates/openshell-sandbox/src/boundary_exec.rs +++ b/crates/openshell-sandbox/src/boundary_exec.rs @@ -168,7 +168,7 @@ impl LocalBoundaryExec { command.env("SHELL", shell); } for (key, value) in &self.user_environment { - if !key.starts_with("OPENSHELL_") { + if !key.starts_with(crate::process::RESERVED_ENV_PREFIX) { command.env(key, value); } } @@ -184,7 +184,7 @@ impl LocalBoundaryExec { } crate::process::strip_proxy_env_std(&mut command); for (key, value) in &spec.env { - if !key.starts_with("OPENSHELL_") { + if !key.starts_with(crate::process::RESERVED_ENV_PREFIX) { command.env(key, value); } } diff --git a/crates/openshell-sandbox/src/boundary_io.rs b/crates/openshell-sandbox/src/boundary_io.rs index 2a005fd965..120ec9130f 100644 --- a/crates/openshell-sandbox/src/boundary_io.rs +++ b/crates/openshell-sandbox/src/boundary_io.rs @@ -7,7 +7,7 @@ use async_trait::async_trait; use openshell_isolation_interface::contract::{ BackendError, BoundaryDuplexStream, BoundaryLoopbackConnector, LoopbackTarget, }; -use std::collections::HashMap; +use std::collections::{HashMap, HashSet}; use std::sync::atomic::{AtomicU8, Ordering}; use std::sync::{Arc, Mutex}; @@ -231,10 +231,11 @@ impl BoundaryRuntimeState { if self.has_registered_processes() { return true; } + // An unreadable /proc fails closed: processes may remain. #[cfg(target_os = "linux")] - if self.exclusive_pid_namespace && sandbox_owns_process_tree() { - return !owned_processes(&[], true).is_empty(); - } + return owned_processes(&[], self.exclusive_pid_namespace) + .map_or(true, |owned| !owned.is_empty()); + #[cfg(not(target_os = "linux"))] false } @@ -283,7 +284,8 @@ impl BoundaryRuntimeState { // requiring ptrace or a capability. let mut previous = Vec::new(); for _ in 0..4 { - let owned = owned_processes(&roots, self.exclusive_pid_namespace); + let owned = + owned_processes(&roots, self.exclusive_pid_namespace).unwrap_or_default(); for process in &owned { if !roots.contains(&process.pid) { signal_owned_process(*process, signal); @@ -339,12 +341,13 @@ fn sandbox_owns_process_tree() -> bool { } #[cfg(target_os = "linux")] -fn owned_processes(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { +fn owned_processes( + roots: &[u32], + exclusive_pid_namespace: bool, +) -> std::io::Result> { let mut stats = HashMap::new(); - let Ok(entries) = std::fs::read_dir("/proc") else { - return Vec::new(); - }; - for entry in entries.flatten() { + let mut children: HashMap> = HashMap::new(); + for entry in std::fs::read_dir("/proc")?.flatten() { let Some(pid) = entry .file_name() .to_str() @@ -353,47 +356,42 @@ fn owned_processes(roots: &[u32], exclusive_pid_namespace: bool) -> Vec>(); + }); + } + } owned.sort_unstable(); - owned.dedup(); - owned + Ok(owned) } /// Signal one scanned process through a pidfd, after confirming the pidfd diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index 5593627653..b1833eadda 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -340,10 +340,8 @@ mod linux { .to_string(), ); } - // The TCP control listener rejects loopback-interface ingress - // only when it serves a supervisor in another network namespace. - // A loopback listener would be reachable from workload sockets - // that the broker does not track, such as natively accepted ones. + // Workload sockets share the loopback interface with a loopback + // listener. BoundaryListenerConfig::TlsTcp { address, .. } if address.ip().to_canonical().is_loopback() => { @@ -3240,15 +3238,10 @@ mod linux { } } - /// Bind the TCP control listener. - /// - /// A listener on a non-loopback address serves a supervisor in another - /// network namespace, so it rejects all loopback-interface ingress - /// before it starts listening. Workload sockets are bound to loopback - /// and natively accepted ones are not registered with the broker, so - /// this standing filter, not the broker's port reservation, keeps the - /// workload from reaching the control endpoint through loopback or - /// the pod's own address. + /// Bind the TCP control listener, dropping loopback-interface ingress + /// before it listens so workload sockets cannot reach it through + /// loopback or the pod's own address. Configuration rejects loopback + /// addresses; tests bind them without the filter. fn bind_tcp(address: std::net::SocketAddr) -> io::Result { let socket = socket2::Socket::new( socket2::Domain::for_address(address), diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 0a64fd99d7..23dd2cf8da 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -40,11 +40,6 @@ const DNS_RELAY_ADDRESS: SocketAddr = SocketAddr::V4(std::net::SocketAddrV4::new const RELAY_CONNECT_TIMEOUT: Duration = Duration::from_secs(5); const NETWORK_DECISION_TIMEOUT: Duration = Duration::from_secs(30); -#[cfg(test)] -const PANIC_SENTINEL_LEVEL: i32 = libc::IPPROTO_TCP; -#[cfg(test)] -const PANIC_SENTINEL_OPTION: i32 = 0x7fff_fffe; - fn retry_notification_receive(error: &io::Error) -> bool { error.kind() == io::ErrorKind::Interrupted || error.raw_os_error() == Some(libc::ENOENT) } @@ -557,7 +552,7 @@ fn dispatch_notification( &listener, notification, std::process::id(), - queues.workload_frozen.load(Ordering::Acquire), + &queues.workload_frozen, ); } if matches!( @@ -568,7 +563,7 @@ fn dispatch_notification( &listener, notification, std::process::id(), - queues.workload_frozen.load(Ordering::Acquire), + &queues.workload_frozen, ); } if syscall == libc::SYS_socket { @@ -599,11 +594,6 @@ fn dispatch_notification( .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; let option = i32::try_from(notification.args[2]) .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; - #[cfg(test)] - assert!( - !(level == PANIC_SENTINEL_LEVEL && option == PANIC_SENTINEL_OPTION), - "test-only setsockopt panic sentinel" - ); if socket_option_is_denied(level, option) { return Err(io::Error::from_raw_os_error(libc::EPERM)); } @@ -616,7 +606,7 @@ fn dispatch_notification( /// that another thread cannot replace before the kernel reads them. /// /// Interface-selection options could redirect or unpin a socket's loopback -/// device binding. The kernel already refuses to change an existing binding +/// device binding; the others enable Fast Open or change the address family. The kernel already refuses to change an existing binding /// without `CAP_NET_RAW`; denying them here keeps confinement independent of /// the capability state of the namespace that owns the network namespace. fn socket_option_is_denied(level: i32, option: i32) -> bool { @@ -997,7 +987,9 @@ fn connect_socket( /// nonblocking socket gets the native `EINPROGRESS` and the kernel completes /// the handshake on the shared socket; a blocking socket waits on a bounded /// worker thread. A slow or full local listener therefore cannot stall -/// mediation of unrelated syscalls. +/// mediation of unrelated syscalls. A nonblocking socket is recorded as +/// connected once the handshake starts, so a repeated `connect` reports +/// `EISCONN` even while the handshake is still in progress. fn connect_local_tcp( registry: &Arc>, listener: &Arc, @@ -1046,7 +1038,17 @@ fn connect_local_tcp( .name("openshell-local-connect".to_string()) .spawn(move || { let _slot = slot; - let result = connect_exact(connector.as_raw_fd(), destination); + // The duplicate shares the workload's blocking open file; wait + // natively without changing its flags. + let result = with_sockaddr(destination, |pointer, length| { + // SAFETY: pointer/length describe a live sockaddr and the + // connector is a live duplicate of the workload's socket. + if unsafe { libc::connect(connector.as_raw_fd(), pointer, length) } == 0 { + Ok(()) + } else { + Err(io::Error::last_os_error()) + } + }); if result.is_ok() { commit_local_connect( ®istry, @@ -1099,7 +1101,6 @@ fn repeated_connect_outcome( ) -> Option { match state { SocketState::Created | SocketState::Bound { .. } | SocketState::Listening { .. } => None, - SocketState::Failed { errno } => Some(*errno), // UDP connect replaces the association; repeating the same one succeeds. SocketState::DnsUdp { relay } if *relay == destination => Some(0), SocketState::Local { peer } if kind == InetKind::DnsUdp && *peer == destination => Some(0), @@ -2253,116 +2254,6 @@ mod tests { /// Bound device name and the errno from an attempted rebind. type ConfinementObservation = (Option>, Option); - #[test] - fn direct_syscall_accept_and_getpeername_report_native_peer() { - // Static binaries and Go issue raw syscalls without a libc wrapper. - let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() - .expect("start workload launcher"); - let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); - let (ready_tx, ready_rx) = std::sync::mpsc::sync_channel(1); - let workload = std::thread::spawn(move || { - launcher - .execute(move || -> io::Result<(SocketAddr, SocketAddr)> { - let listener = TcpListener::bind("127.0.0.1:0")?; - ready_tx - .send(listener.local_addr()?) - .map_err(|_| io::Error::other("test client disappeared"))?; - // rustix issues accept4 and getpeername as raw syscalls. - let (accepted, accepted_peer) = - rustix::net::acceptfrom_with(&listener, rustix::net::SocketFlags::CLOEXEC)?; - let as_socket_addr = |address: Option| { - address - .and_then(|address| SocketAddr::try_from(address).ok()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EAFNOSUPPORT)) - }; - let accepted_peer = as_socket_addr(accepted_peer)?; - let peer = as_socket_addr(rustix::net::getpeername(&accepted)?)?; - Ok((accepted_peer, peer)) - }) - .expect("launcher result") - }); - let address = ready_rx.recv().expect("workload listener ready"); - let client = TcpStream::connect(address).expect("connect loopback client"); - let (accepted_peer, peer) = workload - .join() - .expect("join workload") - .expect("direct-syscall accept"); - assert_eq!(accepted_peer, client.local_addr().unwrap()); - assert_eq!(peer, accepted_peer); - } - - #[test] - #[ignore = "requires OPENSHELL_STATIC_SERVER pointing at a static test server"] - fn static_server_accepts_under_workload_filter() { - // The server listens on argv[1], prints "ready", accepts one - // connection, and writes the peer address it observed. - use std::io::BufRead as _; - let server = std::env::var("OPENSHELL_STATIC_SERVER").expect("OPENSHELL_STATIC_SERVER"); - let address = { - let probe = TcpListener::bind("127.0.0.1:0").unwrap(); - probe.local_addr().unwrap() - }; - let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() - .expect("start workload launcher"); - let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); - let connections: usize = std::env::var("OPENSHELL_STATIC_SERVER_CONNECTIONS") - .map_or(1, |value| value.parse().expect("connection count")); - let mut child = launcher - .execute(move || { - std::process::Command::new(server) - .arg(address.to_string()) - .arg(connections.to_string()) - .stdout(std::process::Stdio::piped()) - .stderr(std::process::Stdio::piped()) - .spawn() - }) - .unwrap() - .expect("spawn static server under the workload filter"); - let mut stdout = io::BufReader::new(child.stdout.take().unwrap()); - let mut line = String::new(); - stdout.read_line(&mut line).unwrap(); - assert_eq!(line.trim(), "ready", "server did not start"); - // Earlier connections measure accept latency through the workload - // filter; the last one also verifies the observed peer below. - let mut latencies = Vec::with_capacity(connections); - let started = Instant::now(); - for _ in 1..connections { - let begin = Instant::now(); - let mut client = TcpStream::connect(address).expect("connect to static server"); - let mut reply = String::new(); - client.read_to_string(&mut reply).unwrap(); - latencies.push(begin.elapsed()); - } - let begin = Instant::now(); - let mut client = TcpStream::connect(address).expect("connect to static server"); - let mut reply = String::new(); - client.read_to_string(&mut reply).unwrap(); - latencies.push(begin.elapsed()); - let total = started.elapsed(); - latencies.sort_unstable(); - let percentile = |p: usize| latencies[(latencies.len() - 1) * p / 100]; - eprintln!( - "static server: {connections} connections in {total:?}; p50 {:?} p99 {:?} max {:?}", - percentile(50), - percentile(99), - latencies[latencies.len() - 1] - ); - let status = child.wait().unwrap(); - let mut stderr = String::new(); - child - .stderr - .take() - .unwrap() - .read_to_string(&mut stderr) - .unwrap(); - assert!(status.success(), "static server failed: {stderr}"); - assert_eq!( - reply.trim(), - client.local_addr().unwrap().to_string(), - "server observed the wrong peer" - ); - } - #[test] fn workload_sockets_are_bound_to_loopback_and_cannot_be_rebound() { let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() @@ -2508,14 +2399,6 @@ mod tests { ), (SocketState::DnsUdp { relay }, InetKind::DnsUdp, other, None), (SocketState::Local { peer }, InetKind::DnsUdp, peer, Some(0)), - ( - SocketState::Failed { - errno: libc::ECONNRESET, - }, - InetKind::Tcp, - peer, - Some(libc::ECONNRESET), - ), ] { assert_eq!( repeated_connect_outcome(&state, kind, destination), @@ -2526,10 +2409,10 @@ mod tests { } #[test] - fn restarted_bind_and_connect_are_answered_consistently() { - // Without killable notification waits a signal can restart a - // syscall the broker already completed. A repeat must not fail on - // the broker's released pre-connect descriptor. + fn repeated_bind_and_connect_after_completion() { + // A signal can restart a syscall the broker already completed. A + // repeated connect reports EISCONN; a repeated bind of the same + // address succeeds, unlike a native EINVAL, so a restart is safe. let service = TcpListener::bind("127.0.0.1:0").unwrap(); let address = service.local_addr().unwrap(); let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() @@ -2652,7 +2535,9 @@ mod tests { .connect(&address.into()) .err() .and_then(|error| error.raw_os_error()); - drop(pending.join()); + // Leave the pending connect running; closing the listener when + // the test ends resets it. + drop(pending); drop(filled); (blocking_wait, nonblocking_result) }) @@ -2664,43 +2549,6 @@ mod tests { assert_eq!(nonblocking_result, Some(libc::EINPROGRESS)); } - #[test] - fn a_panicking_handler_does_not_kill_the_broker() { - // A handler panic must fail only that syscall, not hang every later - // mediated syscall by silently killing the broker thread. setsockopt - // is routed through a hook that panics in test builds on a sentinel - // option, standing in for an unexpected handler bug. - let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() - .expect("start workload launcher"); - let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); - let (panicked, after) = launcher - .execute(|| { - let socket = - socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None) - .unwrap(); - // SAFETY: scalar setsockopt with the test-only panic sentinel. - let value = 1_i32; - let panicked = unsafe { - libc::setsockopt( - socket.as_raw_fd(), - PANIC_SENTINEL_LEVEL, - PANIC_SENTINEL_OPTION, - (&raw const value).cast(), - libc::socklen_t::try_from(size_of::()).unwrap(), - ) - }; - // A later mediated syscall still completes, proving the broker - // survived the panic. - let after = socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::DGRAM, None) - .map(drop) - .map_err(|error| error.raw_os_error()); - (panicked, after) - }) - .expect("launcher result"); - assert_eq!(panicked, -1, "panicking syscall must fail"); - assert_eq!(after, Ok(()), "broker must keep mediating after a panic"); - } - #[test] fn workload_cannot_bind_a_non_loopback_source_address() { // Workload sockets can only present loopback source addresses. A @@ -2861,8 +2709,7 @@ mod tests { #[test] fn udp_dns_after_connect_sends_with_sendmmsg() { // glibc connects the resolver socket to the nameserver, then sends A - // and AAAA together with sendmmsg and no destination. sendmmsg is - // mediated, so the broker must still hold its copy of the socket. + // and AAAA together with sendmmsg and no destination. let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() .expect("start workload launcher"); let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); @@ -2908,17 +2755,7 @@ mod tests { .build() .expect("test runtime"); for _ in 0..2 { - // Bound the wait so a broken send path fails instead of hanging. - let Ok(query) = runtime.block_on(async { - tokio::time::timeout(Duration::from_secs(5), broker.accept_dns()).await - }) else { - let error = client - .join() - .expect("join client") - .expect_err("client must fail when no query arrives"); - panic!("DNS query never reached the relay: {error}"); - }; - let query = query.expect("DNS query"); + let query = runtime.block_on(broker.accept_dns()).expect("DNS query"); let response = if query.request == b"dns-query-a" { b"dns-response-a".to_vec() } else if query.request == b"dns-query-aaaa" { @@ -2936,9 +2773,9 @@ mod tests { #[test] fn dns_send_with_ancillary_data_is_refused() { - // Ancillary control data (e.g. IP_PKTINFO) can carry a per-message - // routing override. The broker sends from its own socket without - // control messages, so a workload that supplies them is refused. + // Ancillary data can carry a per-message routing override such as + // IP_PKTINFO. The kernel performs mediated DNS sends, so control data + // is refused. let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() .expect("start workload launcher"); let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); @@ -2952,7 +2789,7 @@ mod tests { iov_base: payload.as_ptr().cast_mut().cast(), iov_len: payload.len(), }; - // One IP_PKTINFO control message. + // Any non-empty control buffer is refused. let mut control = [0_u8; 32]; let header = libc::msghdr { msg_name: native.as_ptr().cast_mut().cast(), diff --git a/crates/openshell-sandbox/src/process.rs b/crates/openshell-sandbox/src/process.rs index 20ae60d5b6..20812ce315 100644 --- a/crates/openshell-sandbox/src/process.rs +++ b/crates/openshell-sandbox/src/process.rs @@ -106,7 +106,7 @@ pub(crate) fn ca_runtime_read_only_paths(ca_paths: Option<&(PathBuf, PathBuf)>) } /// Prefix of environment variable names reserved for `OpenShell`. -const RESERVED_ENV_PREFIX: &str = "OPENSHELL_"; +pub(crate) const RESERVED_ENV_PREFIX: &str = "OPENSHELL_"; const SUPERVISOR_ONLY_ENV_VARS: &[&str] = &[ openshell_core::sandbox_env::OCI_IMAGE_USER, @@ -213,7 +213,7 @@ fn apply_canonical_process_environment( // reserved OPENSHELL_ namespace inherited from the sandbox itself, which // carries its own control state (for example the serialized user // environment and log level), then restore the one marker the workload is - // meant to see. Declared variables cannot use this namespace. + // meant to see. The gateway rejects declared variables in this namespace. for (key, _) in std::env::vars_os() { if key .to_str() @@ -468,9 +468,7 @@ impl ProcessHandle { provider_env: &HashMap, ) -> Result { let mut cmd = Command::new(program); - cmd.args(args) - .kill_on_drop(true) - .env(openshell_core::sandbox_env::SANDBOX, "1"); + cmd.args(args).kill_on_drop(true); let mut pty_master = None; let mut terminal_slave_fd = None; @@ -633,9 +631,7 @@ impl ProcessHandle { provider_env: &HashMap, ) -> Result { let mut cmd = Command::new(program); - cmd.args(args) - .kill_on_drop(true) - .env(openshell_core::sandbox_env::SANDBOX, "1"); + cmd.args(args).kill_on_drop(true); let mut pty_master = None; let mut terminal_slave_fd = None; diff --git a/crates/openshell-sandbox/src/provider_files.rs b/crates/openshell-sandbox/src/provider_files.rs index e19134a74f..384eb6b31f 100644 --- a/crates/openshell-sandbox/src/provider_files.rs +++ b/crates/openshell-sandbox/src/provider_files.rs @@ -206,15 +206,13 @@ fn handle_thread_comm_open( if thread_group_of(target) != Some(caller_group) { return Ok(false); } - if flags - & (libc::O_CREAT - | libc::O_EXCL - | libc::O_TRUNC - | libc::O_TMPFILE - | libc::O_DIRECTORY - | libc::O_PATH) - != 0 - { + // A shell redirect opens with O_CREAT|O_TRUNC; both are no-ops on an + // existing comm file. O_EXCL fails as it would natively. + if flags & libc::O_EXCL != 0 { + listener.respond_errno(notification.id, libc::EEXIST)?; + return Ok(true); + } + if flags & (libc::O_TMPFILE | libc::O_DIRECTORY | libc::O_PATH) != 0 { listener.respond_errno(notification.id, libc::EINVAL)?; return Ok(true); } @@ -327,7 +325,7 @@ fn sealed_memfd(content: &[u8]) -> io::Result { mod tests { use super::{ProviderFiles, comm_target, sealed_memfd}; use std::collections::HashMap; - use std::io::{Read as _, Write as _}; + use std::io::Read as _; use std::os::fd::AsRawFd as _; use std::os::unix::fs::PermissionsExt as _; @@ -351,55 +349,6 @@ mod tests { } } - #[test] - fn workload_thread_rename_through_proc_is_served() { - // Under the workload filter, a thread rename via its own comm file - // is served by the broker and takes effect. - const CHILD_MARKER: &str = "OPENSHELL_THREAD_COMM_CHILD"; - if std::env::var_os(CHILD_MARKER).is_some() { - let (sender, receiver) = std::sync::mpsc::channel(); - let (done, wait) = std::sync::mpsc::channel::<()>(); - let worker = std::thread::spawn(move || { - sender.send(nix::unistd::gettid().as_raw()).unwrap(); - let _ = wait.recv(); - }); - let tid = receiver.recv().unwrap(); - let path = format!("/proc/self/task/{tid}/comm"); - let mut file = std::fs::OpenOptions::new() - .read(true) - .write(true) - .open(&path) - .expect("open own thread comm"); - file.write_all(b"renamed").expect("rename thread"); - let name = std::fs::read_to_string(&path).unwrap(); - done.send(()).unwrap(); - worker.join().unwrap(); - assert_eq!(name.trim(), "renamed"); - return; - } - let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() - .expect("start workload launcher"); - let _broker = crate::network_broker::NetworkBroker::start_for_test(listener) - .expect("start network broker"); - let status = launcher - .execute(|| { - std::process::Command::new(std::env::current_exe().unwrap()) - .args([ - "--exact", - "provider_files::tests::workload_thread_rename_through_proc_is_served", - "--nocapture", - ]) - .env(CHILD_MARKER, "1") - .status() - }) - .unwrap() - .expect("run workload child"); - assert!( - status.success(), - "thread rename under the workload filter failed" - ); - } - #[test] fn paths_cannot_escape_the_managed_tree() { let valid = "/run/openshell/providers/acme/client.toml"; diff --git a/docs/about/support-matrix.mdx b/docs/about/support-matrix.mdx index 23d607eb87..c62d401183 100644 --- a/docs/about/support-matrix.mdx +++ b/docs/about/support-matrix.mdx @@ -171,7 +171,7 @@ when it runs inside a container or microVM: | [Landlock LSM](https://docs.kernel.org/security/landlock.html) | Required | ABI 3 or newer, introduced in Linux 6.2, with Landlock enabled. The mandatory baseline protects private channel and bootstrap files, including against truncation. A filesystem policy's `best_effort` setting never disables this baseline. | | seccomp | Required | Nested user-notification filters and atomic `SECCOMP_IOCTL_NOTIF_ADDFD` with `SECCOMP_ADDFD_FLAG_SEND`, usable under the runtime's existing seccomp profile without added capabilities. The sandbox actively probes these operations before admitting the workload. | | Task-memory access | Required | The non-dumpable broker must be able to read a same-UID, dumpable workload child's memory through `process_vm_readv` or `/proc//mem`. The broker never writes workload memory. The sandbox actively probes the production parent-to-child topology before admitting the workload. | -| Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE` before handing it to the workload, without added capabilities (Linux 5.14+). Accepted sockets inherit the binding, so `accept` runs natively and only admits clients in the sandbox's own network namespace. Such a client that is not a workload process may present a non-loopback source address. No broker limit applies to accepted connections; each process's open-file limit (`RLIMIT_NOFILE`), the runtime PID limit, and the sandbox memory limit bound them. The sandbox actively probes that the binding installs, cannot be cleared, and is inherited before admitting the workload. | +| Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE`, without added capabilities. Accepted sockets inherit the binding, so workloads accept connections only from inside the sandbox network namespace. The sandbox probes this before admitting the workload. | A kernel version alone does not establish support. A disabled Landlock LSM or a runtime profile that blocks the required seccomp operations causes launch to diff --git a/docs/kubernetes/openshift.mdx b/docs/kubernetes/openshift.mdx index 191ad3521a..7e7887f092 100644 --- a/docs/kubernetes/openshift.mdx +++ b/docs/kubernetes/openshift.mdx @@ -23,8 +23,7 @@ OpenShell fails sandbox startup when either capability-free runtime probe fails. OpenShell requires OpenShift 4.19 or later. The RHCOS kernels in OpenShift 4.16 through 4.18 (RHEL 9.4, 5.14.0-427) are built without Landlock, so sandbox -startup fails its Landlock probe on those releases. On supported releases the -sandbox behaves the same as on newer upstream kernels. +startup fails its Landlock probe on those releases. ## Prerequisites From 65d8d56d7963288203a175e190924a00bb78c730 Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 22:58:24 -0700 Subject: [PATCH 17/18] fix(cli): wait for input when piped exec stdin is nonblocking Processes that inherit the same stdin share one open file description, so any of them can make it nonblocking for all. `sandbox exec` then read no input yet, got EAGAIN, and failed with "Resource temporarily unavailable (os error 11)". Parallel e2e tests inherit the runner's stdin, which made credential_gating fail intermittently. The stdin reader now blocks in poll() until input or end of file arrives instead of treating EAGAIN as fatal, and it leaves the shared descriptor's flags alone. The e2e exec helper also stops inheriting the runner's stdin, since its commands take no input. Signed-off-by: Drew Newberry --- crates/openshell-cli/Cargo.toml | 2 +- crates/openshell-cli/src/run.rs | 72 ++++++++++++++++++++++++++++++++- e2e/rust/src/harness/sandbox.rs | 6 ++- 3 files changed, 77 insertions(+), 3 deletions(-) diff --git a/crates/openshell-cli/Cargo.toml b/crates/openshell-cli/Cargo.toml index f097bb856f..5c92eefe5b 100644 --- a/crates/openshell-cli/Cargo.toml +++ b/crates/openshell-cli/Cargo.toml @@ -88,7 +88,7 @@ tracing-subscriber = { workspace = true } workspace = true [target.'cfg(unix)'.dependencies] -nix = { workspace = true } +nix = { workspace = true, features = ["poll"] } [dev-dependencies] # Tests import the example profiles from providers/ the way an operator diff --git a/crates/openshell-cli/src/run.rs b/crates/openshell-cli/src/run.rs index 028b10dfdd..7ba04d9702 100644 --- a/crates/openshell-cli/src/run.rs +++ b/crates/openshell-cli/src/run.rs @@ -1868,7 +1868,7 @@ enum PipedStdin { /// never waits on a thread parked in `read(2)`. The thread exits at EOF, on a /// read error, or when the receiver is dropped. fn spawn_piped_stdin_reader( - mut reader: impl Read + Send + 'static, + mut reader: impl PipedInput, ) -> tokio::sync::mpsc::Receiver>> { let (tx, rx) = tokio::sync::mpsc::channel::>>(64); std::thread::spawn(move || { @@ -1877,6 +1877,16 @@ fn spawn_piped_stdin_reader( match reader.read(&mut buf) { Ok(0) => return, Err(error) if error.kind() == ErrorKind::Interrupted => {} + // Processes that inherit the same stdin share its open file + // description, so another process may have made it + // nonblocking. Wait for input instead of failing. + #[cfg(unix)] + Err(error) if error.kind() == ErrorKind::WouldBlock => { + if let Err(error) = wait_until_readable(&reader) { + let _ = tx.blocking_send(Err(error)); + return; + } + } Err(error) => { let _ = tx.blocking_send(Err(error)); return; @@ -1892,6 +1902,30 @@ fn spawn_piped_stdin_reader( rx } +/// Input the piped-stdin reader accepts. On Unix it must expose a descriptor +/// so a nonblocking stream can be waited on. +#[cfg(unix)] +trait PipedInput: Read + std::os::fd::AsFd + Send + 'static {} +#[cfg(unix)] +impl PipedInput for T {} +#[cfg(not(unix))] +trait PipedInput: Read + Send + 'static {} +#[cfg(not(unix))] +impl PipedInput for T {} + +/// Block until `reader` has input or reaches end of file. +#[cfg(unix)] +fn wait_until_readable(reader: &impl std::os::fd::AsFd) -> std::io::Result<()> { + use nix::poll::{PollFd, PollFlags, PollTimeout, poll}; + let mut fds = [PollFd::new(reader.as_fd(), PollFlags::POLLIN)]; + loop { + match poll(&mut fds, PollTimeout::NONE) { + Err(nix::errno::Errno::EINTR) => {} + result => return result.map(drop).map_err(std::io::Error::from), + } + } +} + /// Collect piped stdin until EOF or until `grace` elapses, whichever comes /// first. Input beyond `limit` bytes is rejected with the upload hint. async fn collect_piped_stdin( @@ -8585,6 +8619,42 @@ mod tests { ); } + #[cfg(unix)] + #[test] + fn piped_stdin_left_nonblocking_by_another_process_still_streams() { + // Processes that inherit the same stdin share one open file + // description, so any of them can make it nonblocking for all. A + // read with no input yet then fails with EAGAIN instead of waiting. + use std::os::fd::AsRawFd as _; + let (reader, mut writer) = std::io::pipe().expect("pipe"); + nix::fcntl::fcntl( + reader.as_raw_fd(), + nix::fcntl::FcntlArg::F_SETFL(nix::fcntl::OFlag::O_NONBLOCK), + ) + .expect("make the shared pipe nonblocking"); + let runtime = exec_stdin_runtime(); + let collected = runtime.block_on(super::collect_piped_stdin( + super::spawn_piped_stdin_reader(reader), + Duration::from_millis(100), + super::MAX_EXEC_STDIN_BYTES, + )); + let super::PipedStdin::Open { prefix, mut rest } = collected.expect("collect") else { + panic!("an open pipe must start the command before EOF"); + }; + assert!(prefix.is_empty()); + writer.write_all(b"late").unwrap(); + drop(writer); + let next = runtime + .block_on(rest.recv()) + .expect("late chunk") + .expect("read"); + assert_eq!(next, b"late"); + assert!( + runtime.block_on(rest.recv()).is_none(), + "EOF closes the channel" + ); + } + #[test] fn piped_stdin_over_the_limit_is_rejected_with_the_upload_hint() { let (reader, mut writer) = std::io::pipe().expect("pipe"); diff --git a/e2e/rust/src/harness/sandbox.rs b/e2e/rust/src/harness/sandbox.rs index 9922db02b4..0ebfde9081 100644 --- a/e2e/rust/src/harness/sandbox.rs +++ b/e2e/rust/src/harness/sandbox.rs @@ -618,7 +618,11 @@ impl SandboxGuard { for arg in argv { cmd.arg(arg); } - cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); + // Never share the test runner's stdin: parallel test processes share + // its open file description, and the command needs no input. + cmd.stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); let output = cmd .output() From 0325f86368d4532329f04564c07c24d261e6340b Mon Sep 17 00:00:00 2001 From: Drew Newberry Date: Fri, 2 Oct 2026 23:21:50 -0700 Subject: [PATCH 18/18] fix(sandbox): fix macOS lint and older-glibc test linking Import HashSet only in the Linux-only process scan, and call gettid through the raw syscall in a test, since the libc wrapper needs glibc 2.30 or newer. Signed-off-by: Drew Newberry --- crates/openshell-sandbox/src/boundary_io.rs | 4 ++-- crates/openshell-sandbox/src/network_broker.rs | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/crates/openshell-sandbox/src/boundary_io.rs b/crates/openshell-sandbox/src/boundary_io.rs index 120ec9130f..1889052421 100644 --- a/crates/openshell-sandbox/src/boundary_io.rs +++ b/crates/openshell-sandbox/src/boundary_io.rs @@ -7,7 +7,7 @@ use async_trait::async_trait; use openshell_isolation_interface::contract::{ BackendError, BoundaryDuplexStream, BoundaryLoopbackConnector, LoopbackTarget, }; -use std::collections::{HashMap, HashSet}; +use std::collections::HashMap; use std::sync::atomic::{AtomicU8, Ordering}; use std::sync::{Arc, Mutex}; @@ -371,7 +371,7 @@ fn owned_processes( } else { roots.to_vec() }; - let mut visited = HashSet::new(); + let mut visited = std::collections::HashSet::new(); let mut owned = Vec::new(); while let Some(pid) = pending.pop() { if !visited.insert(pid) { diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index 23dd2cf8da..8988ae3328 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -2455,7 +2455,7 @@ mod tests { libc::syscall( libc::SYS_tgkill, libc::getpid(), - libc::gettid(), + libc::syscall(libc::SYS_gettid), libc::SIGCONT, ) };