diff --git a/Cargo.lock b/Cargo.lock index fc0b173702..3fbee07a53 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4413,6 +4413,7 @@ dependencies = [ "serde", "serde_json", "sha2 0.10.9", + "socket2", "tokio", ] diff --git a/crates/openshell-cli/Cargo.toml b/crates/openshell-cli/Cargo.toml index f097bb856f..5c92eefe5b 100644 --- a/crates/openshell-cli/Cargo.toml +++ b/crates/openshell-cli/Cargo.toml @@ -88,7 +88,7 @@ tracing-subscriber = { workspace = true } workspace = true [target.'cfg(unix)'.dependencies] -nix = { workspace = true } +nix = { workspace = true, features = ["poll"] } [dev-dependencies] # Tests import the example profiles from providers/ the way an operator diff --git a/crates/openshell-cli/src/run.rs b/crates/openshell-cli/src/run.rs index 028b10dfdd..7ba04d9702 100644 --- a/crates/openshell-cli/src/run.rs +++ b/crates/openshell-cli/src/run.rs @@ -1868,7 +1868,7 @@ enum PipedStdin { /// never waits on a thread parked in `read(2)`. The thread exits at EOF, on a /// read error, or when the receiver is dropped. fn spawn_piped_stdin_reader( - mut reader: impl Read + Send + 'static, + mut reader: impl PipedInput, ) -> tokio::sync::mpsc::Receiver>> { let (tx, rx) = tokio::sync::mpsc::channel::>>(64); std::thread::spawn(move || { @@ -1877,6 +1877,16 @@ fn spawn_piped_stdin_reader( match reader.read(&mut buf) { Ok(0) => return, Err(error) if error.kind() == ErrorKind::Interrupted => {} + // Processes that inherit the same stdin share its open file + // description, so another process may have made it + // nonblocking. Wait for input instead of failing. + #[cfg(unix)] + Err(error) if error.kind() == ErrorKind::WouldBlock => { + if let Err(error) = wait_until_readable(&reader) { + let _ = tx.blocking_send(Err(error)); + return; + } + } Err(error) => { let _ = tx.blocking_send(Err(error)); return; @@ -1892,6 +1902,30 @@ fn spawn_piped_stdin_reader( rx } +/// Input the piped-stdin reader accepts. On Unix it must expose a descriptor +/// so a nonblocking stream can be waited on. +#[cfg(unix)] +trait PipedInput: Read + std::os::fd::AsFd + Send + 'static {} +#[cfg(unix)] +impl PipedInput for T {} +#[cfg(not(unix))] +trait PipedInput: Read + Send + 'static {} +#[cfg(not(unix))] +impl PipedInput for T {} + +/// Block until `reader` has input or reaches end of file. +#[cfg(unix)] +fn wait_until_readable(reader: &impl std::os::fd::AsFd) -> std::io::Result<()> { + use nix::poll::{PollFd, PollFlags, PollTimeout, poll}; + let mut fds = [PollFd::new(reader.as_fd(), PollFlags::POLLIN)]; + loop { + match poll(&mut fds, PollTimeout::NONE) { + Err(nix::errno::Errno::EINTR) => {} + result => return result.map(drop).map_err(std::io::Error::from), + } + } +} + /// Collect piped stdin until EOF or until `grace` elapses, whichever comes /// first. Input beyond `limit` bytes is rejected with the upload hint. async fn collect_piped_stdin( @@ -8585,6 +8619,42 @@ mod tests { ); } + #[cfg(unix)] + #[test] + fn piped_stdin_left_nonblocking_by_another_process_still_streams() { + // Processes that inherit the same stdin share one open file + // description, so any of them can make it nonblocking for all. A + // read with no input yet then fails with EAGAIN instead of waiting. + use std::os::fd::AsRawFd as _; + let (reader, mut writer) = std::io::pipe().expect("pipe"); + nix::fcntl::fcntl( + reader.as_raw_fd(), + nix::fcntl::FcntlArg::F_SETFL(nix::fcntl::OFlag::O_NONBLOCK), + ) + .expect("make the shared pipe nonblocking"); + let runtime = exec_stdin_runtime(); + let collected = runtime.block_on(super::collect_piped_stdin( + super::spawn_piped_stdin_reader(reader), + Duration::from_millis(100), + super::MAX_EXEC_STDIN_BYTES, + )); + let super::PipedStdin::Open { prefix, mut rest } = collected.expect("collect") else { + panic!("an open pipe must start the command before EOF"); + }; + assert!(prefix.is_empty()); + writer.write_all(b"late").unwrap(); + drop(writer); + let next = runtime + .block_on(rest.recv()) + .expect("late chunk") + .expect("read"); + assert_eq!(next, b"late"); + assert!( + runtime.block_on(rest.recv()).is_none(), + "EOF closes the channel" + ); + } + #[test] fn piped_stdin_over_the_limit_is_rejected_with_the_upload_hint() { let (reader, mut writer) = std::io::pipe().expect("pipe"); diff --git a/crates/openshell-isolation-interface/Cargo.toml b/crates/openshell-isolation-interface/Cargo.toml index 0426a9011f..fe3e6f9551 100644 --- a/crates/openshell-isolation-interface/Cargo.toml +++ b/crates/openshell-isolation-interface/Cargo.toml @@ -22,7 +22,8 @@ tokio = { workspace = true } libc = "0.2" [target.'cfg(target_os = "linux")'.dependencies] -rustix = { workspace = true, features = ["fs", "process"] } +rustix = { workspace = true, features = ["fs", "net", "process"] } +socket2 = { workspace = true, features = ["all"] } [dev-dependencies] tokio = { workspace = true } diff --git a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs index 4ba4eefafb..6fc9f67f73 100644 --- a/crates/openshell-isolation-interface/src/linux/child_seccomp.rs +++ b/crates/openshell-isolation-interface/src/linux/child_seccomp.rs @@ -29,6 +29,7 @@ const SECCOMP_DATA_ARGS_OFFSET: u32 = 16; const X32_SYSCALL_BIT: u32 = 0x4000_0000; const CLOSE_RANGE_UNSHARE_FLAG: u32 = 1 << 1; +const CLOSE_RANGE_CLOEXEC_FLAG: u32 = 1 << 2; const F_SETOWN_COMMAND: u32 = 8; const F_SETSIG_COMMAND: u32 = 10; const F_SETOWN_EX_COMMAND: u32 = 15; @@ -53,6 +54,7 @@ impl ChildHardeningProgram { /// The caller must invoke this from the post-fork child after all /// sandbox-wide TSYNC work and the launcher's `NEW_LISTENER` filter. pub fn install(&mut self) -> io::Result<()> { + mark_inherited_descriptors_close_on_exec()?; set_no_new_privileges()?; let len = u16::try_from(self.instructions.len()).map_err(|_| { io::Error::new( @@ -88,13 +90,43 @@ impl ChildHardeningProgram { } } +/// Mark every descriptor above stdio close-on-exec in the post-fork child. +/// +/// Workloads receive INET sockets only through broker injection, which binds +/// them to loopback first. A descriptor the sandbox process inherited from its +/// container runtime, or opened without `O_CLOEXEC`, must never cross `exec` +/// as an unconfined socket. The command's stdio is already installed on 0-2 +/// when `pre_exec` hooks run. This is a single async-signal-safe syscall. +/// +/// # Errors +/// +/// Returns the kernel error; kernels without `CLOSE_RANGE_CLOEXEC` (before +/// Linux 5.11) fail closed. +pub fn mark_inherited_descriptors_close_on_exec() -> io::Result<()> { + // SAFETY: close_range takes scalar arguments and only sets FD_CLOEXEC. + let result = unsafe { + libc::syscall( + libc::SYS_close_range, + 3_u32, + u32::MAX, + CLOSE_RANGE_CLOEXEC_FLAG, + ) + }; + if result < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(()) + } +} + /// Build the same-UID workload self-protection program before `fork`. /// /// `sandbox_tgid` is the sandbox PID as visible from its workload namespace. /// The filter blocks thread-targeting operations that name the trusted sandbox /// leader and blocks process-directed operations with the same target. The -/// ordinary workload listener additionally mediates `kill`, `tkill`, and -/// `rt_sigqueueinfo`: Linux accepts nonleader TIDs for these operations, so a +/// ordinary workload listener additionally mediates `kill`, `tkill`, +/// `rt_sigqueueinfo`, and `SIGCONT` sent with `tgkill` or +/// `rt_tgsigqueueinfo`: Linux accepts nonleader TIDs for these operations, so a /// static TGID comparison alone cannot protect future sandbox worker threads. pub fn prepare(sandbox_tgid: u32) -> io::Result { if sandbox_tgid == 0 { @@ -352,6 +384,48 @@ fn set_no_new_privileges() -> io::Result<()> { mod tests { use super::*; + #[test] + fn inherited_sockets_are_marked_close_on_exec_but_stdio_is_not() { + // The sweep changes every descriptor in the calling process, so run + // it in a fresh copy of this test binary rather than the harness. + const CHILD_MARKER: &str = "OPENSHELL_CLOEXEC_SWEEP_CHILD"; + if std::env::var_os(CHILD_MARKER).is_some() { + let socket = rustix::net::socket( + rustix::net::AddressFamily::INET, + rustix::net::SocketType::STREAM, + None, + ) + .expect("inheritable socket"); + assert!( + !rustix::io::fcntl_getfd(&socket) + .unwrap() + .contains(rustix::io::FdFlags::CLOEXEC) + ); + mark_inherited_descriptors_close_on_exec().expect("sweep descriptors"); + assert!( + rustix::io::fcntl_getfd(&socket) + .unwrap() + .contains(rustix::io::FdFlags::CLOEXEC) + ); + assert!( + !rustix::io::fcntl_getfd(io::stderr()) + .unwrap() + .contains(rustix::io::FdFlags::CLOEXEC) + ); + return; + } + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "linux::child_seccomp::tests::inherited_sockets_are_marked_close_on_exec_but_stdio_is_not", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + .expect("run isolated sweep test"); + assert!(status.success(), "isolated sweep test failed"); + } + #[test] fn rejects_zero_sandbox_tgid() { assert_eq!( diff --git a/crates/openshell-isolation-interface/src/linux/mod.rs b/crates/openshell-isolation-interface/src/linux/mod.rs index bad3d329dc..a15b8b6f43 100644 --- a/crates/openshell-isolation-interface/src/linux/mod.rs +++ b/crates/openshell-isolation-interface/src/linux/mod.rs @@ -11,6 +11,7 @@ pub mod landlock; pub mod proc_fd; pub mod process_signal; pub mod seccomp_notify; +pub mod socket_confinement; pub mod socket_registry; pub mod task_memory; pub mod workload_launcher; diff --git a/crates/openshell-isolation-interface/src/linux/process_signal.rs b/crates/openshell-isolation-interface/src/linux/process_signal.rs index 796f548037..f2addd54dc 100644 --- a/crates/openshell-isolation-interface/src/linux/process_signal.rs +++ b/crates/openshell-isolation-interface/src/linux/process_signal.rs @@ -12,6 +12,7 @@ use std::io; use std::os::fd::{AsRawFd, FromRawFd, OwnedFd}; +use std::sync::atomic::{AtomicBool, Ordering}; use crate::linux::seccomp_notify::{Notification, NotificationListener}; use crate::linux::task_memory; @@ -27,6 +28,7 @@ pub fn mediate_process_signal( listener: &NotificationListener, notification: Notification, sandbox_tgid: u32, + workload_frozen: &AtomicBool, ) -> io::Result<()> { listener.validate_id(notification.id)?; let target = scalar_int(notification.args[0]); @@ -38,6 +40,7 @@ pub fn mediate_process_signal( libc::EINVAL })); } + refuse_resume_while_frozen(signal, workload_frozen)?; let target = u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; let retained = retain_signal_target(target, sandbox_tgid)?; // SAFETY: all-zero siginfo consists of valid integer/pointer fields. A @@ -64,6 +67,7 @@ pub fn mediate_process_signal( _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), }; listener.validate_id(notification.id)?; + refuse_resume_while_frozen(signal, workload_frozen)?; // SAFETY: retained owns a live pidfd; info is null or a complete trusted // copy. The kernel targets that process object, never a reused numeric PID. let result = unsafe { @@ -81,37 +85,62 @@ pub fn mediate_process_signal( listener.respond_value(notification.id, 0) } -/// Continue a positive-target `tkill` only when the target thread belongs to -/// an untrusted workload process rather than the sandbox runtime itself. +/// Continue a thread-directed signal (`tkill`, `tgkill`, +/// `rt_tgsigqueueinfo`) aimed at an untrusted workload thread. /// /// Continuing preserves Linux's thread-directed signal semantics, including -/// the cancellation signal used by musl. A target that exits between the -/// ownership check and continuation can only be reused inside the same PID -/// namespace; the static child filter still rejects the sandbox leader. +/// the cancellation signal used by musl. The static child filter rejects the +/// sandbox leader as a `tgkill`/`rt_tgsigqueueinfo` group, and the kernel +/// rejects a thread outside the named group, so those two are notified only +/// for `SIGCONT`. A `tkill` names a bare thread, so its group is resolved here; +/// a target reused between this check and continuation stays inside the same +/// PID namespace. pub fn mediate_thread_signal( listener: &NotificationListener, notification: Notification, sandbox_tgid: u32, + workload_frozen: &AtomicBool, ) -> io::Result<()> { listener.validate_id(notification.id)?; - let target = scalar_int(notification.args[0]); - let signal = scalar_int(notification.args[1]); - if target <= 0 || !(0..=64).contains(&signal) { - return Err(io::Error::from_raw_os_error(if target <= 0 { - libc::EPERM - } else { - libc::EINVAL - })); - } - let target = u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; - let target_group = thread_group_id(target)?; - if target_group == sandbox_tgid || target_group == 0 { - return Err(io::Error::from_raw_os_error(libc::EPERM)); + let signal = match i64::from(notification.syscall) { + libc::SYS_tkill => { + let target = scalar_int(notification.args[0]); + let signal = scalar_int(notification.args[1]); + if target <= 0 { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } + let target = + u32::try_from(target).map_err(|_| io::Error::from_raw_os_error(libc::ESRCH))?; + let target_group = thread_group_id(target)?; + if target_group == sandbox_tgid || target_group == 0 { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } + signal + } + libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo => scalar_int(notification.args[2]), + _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), + }; + if !(0..=64).contains(&signal) { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); } listener.validate_id(notification.id)?; + refuse_resume_while_frozen(signal, workload_frozen)?; listener.respond_continue(notification.id) } +/// While the boundary has stopped the workload for supervisor recovery, a +/// workload process that was not yet stopped must not resume the others. +/// +/// The flag is read immediately before delivery. The freezer does not wait +/// for in-flight notifications, so a signal already past this check when the +/// freeze begins can still be delivered. +fn refuse_resume_while_frozen(signal: i32, workload_frozen: &AtomicBool) -> io::Result<()> { + if signal == libc::SIGCONT && workload_frozen.load(Ordering::Acquire) { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } + Ok(()) +} + fn scalar_int(value: u64) -> i32 { let bytes = value.to_ne_bytes(); #[cfg(target_endian = "little")] @@ -176,8 +205,13 @@ mod tests { .recv_timeout(std::time::Duration::from_secs(5)) .unwrap(); let notification = listener.receive().unwrap(); - let error = - mediate_process_signal(&listener, notification, std::process::id()).unwrap_err(); + let error = mediate_process_signal( + &listener, + notification, + std::process::id(), + &AtomicBool::new(false), + ) + .unwrap_err(); assert_eq!(error.raw_os_error(), Some(libc::EPERM)); listener .respond_errno(notification.id, libc::EPERM) diff --git a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs index a87dc080d5..5ae24cd27d 100644 --- a/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs +++ b/crates/openshell-isolation-interface/src/linux/seccomp_notify.rs @@ -20,7 +20,6 @@ use std::time::Duration; const SECCOMP_SET_MODE_FILTER: libc::c_uint = 1; const SECCOMP_GET_NOTIF_SIZES: libc::c_uint = 3; const SECCOMP_FILTER_FLAG_NEW_LISTENER: libc::c_ulong = 1 << 3; -const SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV: libc::c_ulong = 1 << 5; const SECCOMP_RET_KILL_PROCESS: u32 = 0x8000_0000; const SECCOMP_RET_USER_NOTIF: u32 = 0x7fc0_0000; @@ -141,8 +140,6 @@ pub struct Notification { /// kernel, outer seccomp profile, and LSM posture. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct NotificationProbeReport { - /// Whether `SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` was accepted. - pub wait_killable_recv: bool, features: NotificationProbeFeatures, } @@ -170,81 +167,24 @@ impl NotificationProbeReport { } } -/// Cancellation posture a listener was installed with. -/// -/// `WAIT_KILLABLE_RECV` (Linux 5.19+) keeps the notified workload thread in a -/// kill-only wait so a non-fatal signal cannot resume the mediated syscall -/// after the broker has validated the notification. A plain listener has no -/// such guarantee, so it runs read-only: the broker must refuse every -/// task-memory *output* write to stay cancellation-safe. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ListenerMode { - /// Modern kernel: `WAIT_KILLABLE_RECV` active; full mediation including - /// task-memory output writes. - Killable, - /// Legacy kernel (< 5.19): plain listener; task-memory output writes are - /// disabled so a resumed syscall cannot race a broker write. - LegacyReadOnly, -} - -impl ListenerMode { - /// Stable identifier for qualification output and diagnostics. - #[must_use] - pub fn as_str(self) -> &'static str { - match self { - Self::Killable => "killable", - Self::LegacyReadOnly => "legacy_read_only", - } - } -} - /// Owned listener returned by `SECCOMP_FILTER_FLAG_NEW_LISTENER`. +/// +/// The listener is installed without `WAIT_KILLABLE_RECV`, so mediation is the +/// same on every kernel. A signal can interrupt a notified syscall, which the +/// kernel then restarts or fails with `EINTR`; broker handlers check the +/// notification is still live before acting and answer a repeated operation +/// as the kernel would. pub struct NotificationListener { fd: OwnedFd, - wait_killable_recv: bool, } impl NotificationListener { - /// Construct a listener from an already-owned notification descriptor in a - /// specific mode. Intended for tests that must exercise the legacy - /// read-only fail-closed paths without a `< 5.19` kernel. - #[must_use] - pub fn from_fd_with_mode(fd: OwnedFd, mode: ListenerMode) -> Self { - Self { - fd, - wait_killable_recv: matches!(mode, ListenerMode::Killable), - } - } - /// Raw listener descriptor for readiness integration and diagnostics. #[must_use] pub fn as_raw_fd(&self) -> RawFd { self.fd.as_raw_fd() } - /// Whether the listener was installed with killable receive waits. - #[must_use] - pub fn wait_killable_recv(&self) -> bool { - self.wait_killable_recv - } - - /// The cancellation mode this listener was installed with. - #[must_use] - pub fn mode(&self) -> ListenerMode { - if self.wait_killable_recv { - ListenerMode::Killable - } else { - ListenerMode::LegacyReadOnly - } - } - - /// Whether broker task-memory output writes are disabled for this listener. - /// True exactly in `LegacyReadOnly` mode (no `WAIT_KILLABLE_RECV`). - #[must_use] - pub fn writes_disabled(&self) -> bool { - matches!(self.mode(), ListenerMode::LegacyReadOnly) - } - /// Receive the next kernel notification. pub fn receive(&self) -> io::Result { let mut raw = RawNotification::default(); @@ -272,34 +212,6 @@ impl NotificationListener { Ok(()) } - /// Write broker-produced output into the notified task's memory, closing - /// the validation-to-write race that a plain listener cannot. - /// - /// In `Killable` mode `WAIT_KILLABLE_RECV` keeps the notified workload - /// thread in a kill-only wait, so a non-fatal signal cannot resume the - /// mediated syscall between `validate_id` and this write. In - /// `LegacyReadOnly` mode (kernels < 5.19) there is no such guarantee: a - /// resumed syscall could repurpose the target buffer while the privileged - /// broker writes through the captured tid and pointer — via `/proc//mem` - /// even into pages the workload has since made read-only. There is no way to - /// close that window without the flag, so this fails closed (`EOPNOTSUPP`) - /// rather than racing. Callers must route every task-memory *output* write - /// through this method; input reads never write workload memory and are - /// unaffected. - pub fn write_task_output( - &self, - id: u64, - tid: u32, - address: u64, - data: &[u8], - ) -> io::Result<()> { - if self.writes_disabled() { - return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); - } - self.validate_id(id)?; - crate::linux::task_memory::write_exact(tid, address, data) - } - /// Return a successful scalar result to the notifying syscall. pub fn respond_value(&self, id: u64, value: i64) -> io::Result<()> { self.validate_id(id)?; @@ -407,22 +319,7 @@ pub fn install_listener(syscalls: &[i64]) -> io::Result { verify_notification_sizes()?; set_no_new_privileges()?; - // WAIT_KILLABLE_RECV (Linux 5.19+) keeps the *notified workload thread* in - // a kill-only wait while the broker services its syscall, so a non-fatal - // signal cannot resume the syscall and repurpose its buffers underneath a - // pending broker write. Kernels older than 5.19 (for example RHEL 9.x / - // 5.14 nodes) reject the flag with EINVAL. Rather than refusing to start - // there, fall back to a plain listener so the sandbox boots; the resulting - // listener records `wait_killable_recv = false`, and the broker then fails - // closed on every task-memory output write (see `write_task_output`) - // instead of racing them. Input mediation is unaffected. - match install_listener_with_flags(syscalls, true) { - Ok(listener) => Ok(listener), - Err(error) if error.raw_os_error() == Some(libc::EINVAL) => { - install_listener_with_flags(syscalls, false) - } - Err(error) => Err(error), - } + install_notification_filter(syscalls) } /// Install the capability-free workload networking listener on the calling @@ -430,7 +327,9 @@ pub fn install_listener(syscalls: &[i64]) -> io::Result { /// /// The filter mediates every syscall that can create, select, or materially /// reconfigure an INET endpoint. Connected `send()`/null-destination -/// `sendto()` retains the audited cBPF fast path. +/// `sendto()` retains the audited cBPF fast path. `accept`/`accept4` run +/// natively: the broker binds every INET socket to loopback before injecting +/// it, and accepted sockets inherit their listener's binding. pub fn install_workload_listener() -> io::Result { #[allow(unused_mut)] // SYS_open is unavailable on some architectures. let mut syscalls = vec![ @@ -438,15 +337,14 @@ pub fn install_workload_listener() -> io::Result { libc::SYS_connect, libc::SYS_bind, libc::SYS_listen, - libc::SYS_accept, - libc::SYS_accept4, libc::SYS_sendto, libc::SYS_sendmsg, libc::SYS_sendmmsg, - libc::SYS_getpeername, libc::SYS_setsockopt, libc::SYS_kill, libc::SYS_tkill, + libc::SYS_tgkill, + libc::SYS_rt_tgsigqueueinfo, libc::SYS_rt_sigqueueinfo, libc::SYS_openat, libc::SYS_openat2, @@ -462,30 +360,28 @@ pub fn install_workload_listener() -> io::Result { /// one dedicated thread and moved to an unfiltered broker thread through an /// in-process channel. pub fn probe_notification_api() -> io::Result { - let wait_killable_recv = probe_scalar_round_trip()?; + probe_scalar_round_trip()?; probe_addfd_send()?; probe_connected_sendto_fast_path()?; Ok(NotificationProbeReport { - wait_killable_recv, features: NotificationProbeFeatures(1 | 2 | 4), }) } -fn probe_scalar_round_trip() -> io::Result { +fn probe_scalar_round_trip() -> io::Result<()> { const PROBE_VALUE: libc::c_long = 0x5a17; let (sender, receiver) = mpsc::sync_channel(1); let launcher = thread::spawn(move || -> io::Result { let listener = install_listener(&[libc::SYS_getppid])?; - let wait_killable = listener.wait_killable_recv(); sender - .send((listener, wait_killable)) + .send(listener) .map_err(|_| io::Error::other("notification broker disappeared"))?; // SAFETY: getppid has no pointer arguments. The installed filter causes // the kernel to block here until the broker validates and responds. Ok(unsafe { libc::syscall(libc::SYS_getppid) }) }); - let (listener, wait_killable) = receiver + let listener = receiver .recv() .map_err(|_| io::Error::other("notification launcher disappeared"))?; let notification = match receive_probe_notification(&listener) { @@ -505,7 +401,7 @@ fn probe_scalar_round_trip() -> io::Result { if observed != PROBE_VALUE { return Err(io::Error::other("seccomp response value was not delivered")); } - Ok(wait_killable) + Ok(()) } fn probe_addfd_send() -> io::Result<()> { @@ -759,10 +655,7 @@ fn receive_probe_notification(listener: &NotificationListener) -> io::Result io::Result { +fn install_notification_filter(syscalls: &[i64]) -> io::Result { let mut program = build_filter(syscalls)?; let length = u16::try_from(program.len()) .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "seccomp filter is too large"))?; @@ -770,12 +663,7 @@ fn install_listener_with_flags( len: length, filter: program.as_mut_ptr(), }; - let flags = SECCOMP_FILTER_FLAG_NEW_LISTENER - | if wait_killable_recv { - SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV - } else { - 0 - }; + let flags = SECCOMP_FILTER_FLAG_NEW_LISTENER; // SAFETY: `fprog` points to a live classic-BPF program for the duration of // the syscall. The returned nonnegative value is a newly owned FD. let result = unsafe { @@ -793,10 +681,7 @@ fn install_listener_with_flags( .map_err(|_| io::Error::other("seccomp listener FD does not fit RawFd"))?; // SAFETY: successful NEW_LISTENER returns one newly owned descriptor. let fd = unsafe { OwnedFd::from_raw_fd(fd) }; - Ok(NotificationListener { - fd, - wait_killable_recv, - }) + Ok(NotificationListener { fd }) } fn build_filter(syscalls: &[i64]) -> io::Result> { @@ -821,6 +706,10 @@ fn build_filter(syscalls: &[i64]) -> io::Result> { append_sendto_filter(&mut program)?; continue; } + if matches!(syscall, libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo) { + append_resume_signal_filter(&mut program, syscall)?; + continue; + } let syscall = u32::try_from(syscall) .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "negative syscall number"))?; program.extend([ @@ -861,6 +750,27 @@ fn append_sendto_filter(program: &mut Vec) -> io::Result<()> Ok(()) } +/// Notify a thread-group signal only when it sends `SIGCONT`, which the broker +/// refuses while the workload is frozen. Every other signal (for example Go's +/// preemption signal) stays in the kernel. +fn append_resume_signal_filter( + program: &mut Vec, + syscall: i64, +) -> io::Result<()> { + let syscall = u32::try_from(syscall) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "negative syscall number"))?; + let resume = u32::try_from(libc::SIGCONT) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "negative signal number"))?; + program.extend([ + jump(BPF_JMP_JEQ_K, syscall, 0, 4), + stmt(BPF_LD_W_ABS, argument_word_offset(2, 0)), + jump(BPF_JMP_JEQ_K, resume, 0, 1), + stmt(BPF_RET_K, SECCOMP_RET_USER_NOTIF), + stmt(BPF_RET_K, SECCOMP_RET_ALLOW), + ]); + Ok(()) +} + const fn argument_word_offset(argument: u32, word: u32) -> u32 { SECCOMP_DATA_ARGS_OFFSET + argument * 8 + word * 4 } @@ -1077,59 +987,10 @@ mod tests { let listener = NotificationListener { // SAFETY: successful dup returned a new owned descriptor. fd: unsafe { OwnedFd::from_raw_fd(duplicated) }, - wait_killable_recv: false, }; let error = listener .respond_errno(1, 0) .expect_err("zero errno must fail"); assert_eq!(error.kind(), io::ErrorKind::InvalidInput); } - - #[test] - fn plain_listener_fails_closed_on_output_write() { - // A LegacyReadOnly listener (kernels < 5.19) must refuse every - // task-memory output write rather than race a resumed syscall. The - // guard short-circuits before touching the descriptor or workload - // memory, so a dup of stderr is a sufficient stand-in. - // SAFETY: dup takes one valid descriptor and returns a new descriptor - // or a negative error without modifying memory. - let duplicated = unsafe { libc::dup(libc::STDERR_FILENO) }; - assert!(duplicated >= 0, "duplicate stderr for validation test"); - // SAFETY: successful dup returned a new owned descriptor. - let listener = NotificationListener::from_fd_with_mode( - unsafe { OwnedFd::from_raw_fd(duplicated) }, - ListenerMode::LegacyReadOnly, - ); - assert!(listener.writes_disabled()); - assert_eq!(listener.mode(), ListenerMode::LegacyReadOnly); - let error = listener - .write_task_output(1, 0, 0, &[0_u8; 4]) - .expect_err("plain listener must reject output writes"); - assert_eq!(error.raw_os_error(), Some(libc::EOPNOTSUPP)); - } - - #[test] - fn real_plain_listener_disables_output_writes() { - // Deliberately install a plain NEW_LISTENER (no WAIT_KILLABLE_RECV) - // even on a modern CI kernel and prove the broker write path fails - // closed on the real listener object. - set_no_new_privileges().expect("no_new_privs for listener install"); - let listener = install_listener_with_flags(&[libc::SYS_getppid], false) - .expect("install plain listener"); - assert_eq!(listener.mode(), ListenerMode::LegacyReadOnly); - assert!(listener.writes_disabled()); - let error = listener - .write_task_output(1, 0, 0, &[0_u8; 4]) - .expect_err("plain listener must reject output writes"); - assert_eq!(error.raw_os_error(), Some(libc::EOPNOTSUPP)); - } - - #[test] - fn killable_listener_enables_output_writes() { - set_no_new_privileges().expect("no_new_privs for listener install"); - let listener = install_listener_with_flags(&[libc::SYS_getppid], true) - .expect("install killable listener"); - assert_eq!(listener.mode(), ListenerMode::Killable); - assert!(!listener.writes_disabled()); - } } diff --git a/crates/openshell-isolation-interface/src/linux/socket_confinement.rs b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs new file mode 100644 index 0000000000..715ce4ca9b --- /dev/null +++ b/crates/openshell-isolation-interface/src/linux/socket_confinement.rs @@ -0,0 +1,293 @@ +// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Standing kernel confinement for workload INET sockets. +//! +//! Every workload INET socket is bound to the loopback device before its +//! descriptor is injected. The binding is kernel state on the socket itself, +//! so it survives `dup`, `fork`, `exec`, `AF_UNSPEC` disconnect, and is +//! inherited by sockets accepted from a confined listener. It restricts both +//! directions: route lookups are pinned to `lo`, and listener/UDP lookup only +//! matches packets that arrive on `lo`. Clearing or changing an existing +//! binding requires `CAP_NET_RAW` in the network namespace's owning user +//! namespace, which the capability-free sandbox and workload do not hold. + +use std::io; +use std::os::fd::AsFd; + +use socket2::{Domain, SockFilter, SockRef, Socket, Type}; + +const LOOPBACK_DEVICE: &[u8] = b"lo"; + +/// Bind `fd` to the loopback device. +/// +/// # Errors +/// +/// Returns the kernel error, including `EPERM` when the socket is already +/// bound to a device. +pub fn confine_to_loopback(fd: impl AsFd) -> io::Result<()> { + SockRef::from(&fd).bind_device(Some(LOOPBACK_DEVICE)) +} + +/// Return the device name `fd` is bound to, or `None` when unbound. +/// +/// # Errors +/// +/// Returns the kernel error from `getsockopt(SO_BINDTODEVICE)`. +pub fn bound_device(fd: impl AsFd) -> io::Result>> { + SockRef::from(&fd).device() +} + +/// Drop TCP/UDP ingress that arrives on the loopback interface. +/// +/// Attach this to a trusted listener whose legitimate clients are never in the +/// same network namespace. Matching the ingress interface rather than the +/// source address also rejects connections to the host's own non-loopback +/// address, which the kernel delivers through loopback. The filter is not +/// locked: the listener descriptor never leaves the trusted sandbox process, +/// which marks every descriptor above stdio close-on-exec before running +/// workload code. +/// +/// # Errors +/// +/// Returns the kernel error when the interface index cannot be resolved or +/// the filter cannot be attached. +pub fn reject_loopback_ingress(fd: impl AsFd) -> io::Result<()> { + let index = rustix::net::netdevice::name_to_index(&fd, "lo")?; + reject_ingress_interface(fd, index) +} + +fn reject_ingress_interface(fd: impl AsFd, index: u32) -> io::Result<()> { + // Ancillary loads use the documented negative offset encoding. + let ifindex_offset = (libc::SKF_AD_OFF + libc::SKF_AD_IFINDEX).cast_unsigned(); + let program = [ + filter( + libc::BPF_LD | libc::BPF_W | libc::BPF_ABS, + 0, + 0, + ifindex_offset, + ), + filter(libc::BPF_JMP | libc::BPF_JEQ | libc::BPF_K, 0, 1, index), + filter(libc::BPF_RET | libc::BPF_K, 0, 0, 0), + filter(libc::BPF_RET | libc::BPF_K, 0, 0, u32::MAX), + ]; + SockRef::from(&fd).attach_filter(&program) +} + +#[allow( + clippy::cast_possible_truncation, + reason = "classic BPF opcodes are 16-bit by definition" +)] +const fn filter(code: u32, jt: u8, jf: u8, k: u32) -> SockFilter { + SockFilter::new(code as u16, jt, jf, k) +} + +/// Actively prove loopback confinement under the current runtime profile. +/// +/// For each supported workload socket type this installs the binding and +/// proves that the sandbox credentials cannot clear or replace it. For IPv4 +/// and IPv6 it proves that a stream accepted from a confined listener inherits +/// the binding and keeps it after an `AF_UNSPEC` disconnect. IPv6 is skipped +/// only when the kernel or namespace does not provide it. +/// +/// # Errors +/// +/// Returns an error describing the first failed property. +pub fn probe_loopback_confinement() -> io::Result<()> { + for (domain, kind) in [ + (Domain::IPV4, Type::STREAM), + (Domain::IPV4, Type::DGRAM), + (Domain::IPV6, Type::STREAM), + (Domain::IPV6, Type::DGRAM), + ] { + let socket = match Socket::new(domain, kind, None) { + Ok(socket) => socket, + Err(error) + if domain == Domain::IPV6 && error.raw_os_error() == Some(libc::EAFNOSUPPORT) => + { + continue; + } + Err(error) => return Err(error), + }; + confine_to_loopback(&socket) + .map_err(|error| probe_error("install loopback binding", &error))?; + probe_binding_is_immutable(&socket)?; + } + probe_accept_inherits_binding() +} + +fn probe_binding_is_immutable(socket: &Socket) -> io::Result<()> { + // `None` requests an unbind; replacing the device takes the same path. + if socket.bind_device(None).is_ok() { + return Err(io::Error::other( + "sandbox credentials can clear a socket device binding", + )); + } + if socket.device()?.as_deref() != Some(LOOPBACK_DEVICE) { + return Err(io::Error::other("socket device binding changed")); + } + Ok(()) +} + +fn probe_accept_inherits_binding() -> io::Result<()> { + for loopback in [ + std::net::IpAddr::V4(std::net::Ipv4Addr::LOCALHOST), + std::net::IpAddr::V6(std::net::Ipv6Addr::LOCALHOST), + ] { + let listener = match std::net::TcpListener::bind((loopback, 0)) { + Ok(listener) => listener, + // Kernels or namespaces without IPv6 have no ::1 to bind. + Err(error) + if loopback.is_ipv6() + && matches!( + error.raw_os_error(), + Some(libc::EAFNOSUPPORT | libc::EADDRNOTAVAIL) + ) => + { + continue; + } + Err(error) => return Err(error), + }; + confine_to_loopback(&listener) + .map_err(|error| probe_error("confine probe listener", &error))?; + let _client = std::net::TcpStream::connect(listener.local_addr()?)?; + let (accepted, _) = listener.accept()?; + if bound_device(&accepted)?.as_deref() != Some(LOOPBACK_DEVICE) { + return Err(io::Error::other( + "accepted socket did not inherit the loopback binding", + )); + } + // Natively accepted sockets are not tracked by the broker, so a + // workload can disconnect and reconnect them. The binding must + // survive that transition. + rustix::net::connect_unspec(&accepted) + .map_err(|error| probe_error("disconnect accepted probe socket", &error.into()))?; + if bound_device(&accepted)?.as_deref() != Some(LOOPBACK_DEVICE) { + return Err(io::Error::other( + "accepted socket lost the loopback binding after disconnect", + )); + } + } + Ok(()) +} + +fn probe_error(context: &str, error: &io::Error) -> io::Error { + io::Error::new(error.kind(), format!("{context}: {error}")) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::{Read as _, Write as _}; + use std::net::{Ipv4Addr, SocketAddr, TcpListener, TcpStream}; + use std::time::Duration; + + fn new_socket(domain: Domain, kind: Type) -> Socket { + Socket::new(domain, kind, None).unwrap() + } + + #[test] + fn active_probe_passes_without_capabilities() { + // Root holds CAP_NET_RAW, which may change a device binding. + if rustix::process::geteuid().is_root() { + return; + } + probe_loopback_confinement().expect("loopback confinement probe"); + } + + #[test] + fn unbound_socket_reports_no_device() { + let socket = new_socket(Domain::IPV4, Type::STREAM); + assert_eq!(bound_device(&socket).unwrap(), None); + } + + #[test] + fn confined_socket_cannot_be_rebound() { + if rustix::process::geteuid().is_root() { + return; + } + let socket = new_socket(Domain::IPV4, Type::DGRAM); + confine_to_loopback(&socket).unwrap(); + assert!(confine_to_loopback(&socket).is_err()); + assert_eq!(bound_device(&socket).unwrap().as_deref(), Some(&b"lo"[..])); + } + + fn connect_with_timeout(address: SocketAddr) -> io::Result { + TcpStream::connect_timeout(&address, Duration::from_millis(300)) + } + + #[test] + fn loopback_ingress_filter_rejects_loopback_connections() { + let listener = TcpListener::bind((Ipv4Addr::UNSPECIFIED, 0)).unwrap(); + reject_loopback_ingress(&listener).unwrap(); + listener.set_nonblocking(true).unwrap(); + let port = listener.local_addr().unwrap().port(); + // Dropped SYNs never complete the handshake. + assert!(connect_with_timeout(SocketAddr::from((Ipv4Addr::LOCALHOST, port))).is_err()); + assert_eq!( + listener.accept().unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + } + + #[test] + fn ingress_filter_admits_other_interfaces() { + // Positive control: the same program keyed to an absent interface + // index must leave loopback traffic untouched. + let listener = TcpListener::bind((Ipv4Addr::LOCALHOST, 0)).unwrap(); + reject_ingress_interface(&listener, u32::MAX).unwrap(); + let mut client = connect_with_timeout(listener.local_addr().unwrap()).unwrap(); + let (mut accepted, _) = listener.accept().unwrap(); + client.write_all(b"ping").unwrap(); + let mut buffer = [0_u8; 4]; + accepted.read_exact(&mut buffer).unwrap(); + assert_eq!(&buffer, b"ping"); + } + + /// Return a local non-loopback address, if this namespace has one. + fn local_non_loopback_address() -> Option { + let probe = std::net::UdpSocket::bind((Ipv4Addr::UNSPECIFIED, 0)).ok()?; + probe.connect((Ipv4Addr::new(192, 0, 2, 1), 9)).ok()?; + match probe.local_addr().ok()?.ip() { + std::net::IpAddr::V4(address) if !address.is_loopback() => Some(address), + _ => None, + } + } + + #[test] + fn confined_listener_peers_are_limited_to_this_network_namespace() { + // Only a same-namespace client can reach a loopback-bound listener; + // connecting to the host's own address is refused. + let Some(local) = local_non_loopback_address() else { + eprintln!("skipping: no non-loopback IPv4 address in this namespace"); + return; + }; + let listener = TcpListener::bind((Ipv4Addr::UNSPECIFIED, 0)).unwrap(); + confine_to_loopback(&listener).unwrap(); + listener.set_nonblocking(true).unwrap(); + let port = listener.local_addr().unwrap().port(); + + let connect_from = |source: Ipv4Addr, destination: Ipv4Addr| { + let client = new_socket(Domain::IPV4, Type::STREAM); + client.bind(&SocketAddr::from((source, 0)).into()).unwrap(); + client + .connect_timeout( + &SocketAddr::from((destination, port)).into(), + Duration::from_millis(300), + ) + .map(|()| client) + }; + + // The host's own address is matched against its real interface. + assert!(connect_from(local, local).is_err()); + assert_eq!( + listener.accept().unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + + let client = connect_from(local, Ipv4Addr::LOCALHOST).expect("same-namespace client"); + let (_, peer) = listener.accept().unwrap(); + assert_eq!(peer, client.local_addr().unwrap().as_socket().unwrap()); + assert_eq!(peer.ip(), std::net::IpAddr::V4(local)); + } +} diff --git a/crates/openshell-isolation-interface/src/linux/socket_registry.rs b/crates/openshell-isolation-interface/src/linux/socket_registry.rs index 7a245d1eac..9754720792 100644 --- a/crates/openshell-isolation-interface/src/linux/socket_registry.rs +++ b/crates/openshell-isolation-interface/src/linux/socket_registry.rs @@ -76,8 +76,6 @@ pub enum SocketState { DnsTcp { relay: SocketAddr }, /// Workload-owned listening socket. Listening { local: SocketAddr }, - /// Stream accepted from a verified local peer. - AcceptedLocal { peer: SocketAddr }, /// A committed relay failed after connection. Failed { errno: i32 }, } @@ -254,9 +252,8 @@ impl SocketRegistry { /// Publish a tentative socket in a caller-proven initial state. /// - /// Accepted sockets are created and classified by the trusted broker, so - /// they enter the registry directly as [`SocketState::AcceptedLocal`] - /// rather than pretending to be unconnected. + /// Used when the trusted broker has already established the socket's + /// state before publication, so the entry never appears unconnected. pub fn commit_with_state( &mut self, tentative: TentativeSocket, diff --git a/crates/openshell-isolation-interface/src/linux/task_memory.rs b/crates/openshell-isolation-interface/src/linux/task_memory.rs index 1b37685b3b..ed34f234e8 100644 --- a/crates/openshell-isolation-interface/src/linux/task_memory.rs +++ b/crates/openshell-isolation-interface/src/linux/task_memory.rs @@ -64,53 +64,6 @@ pub fn read_exact(tid: u32, address: u64, destination: &mut [u8]) -> io::Result< } } -/// Write exactly all of `source` to `address` in `tid`. -/// -/// This is used only for syscall outputs such as `getpeername` and -/// `sendmmsg.msg_len`. Revalidate the notification, task generation, and -/// destination layout immediately before calling it. -pub fn write_exact(tid: u32, address: u64, source: &[u8]) -> io::Result<()> { - validate_request(tid, address, source.len())?; - let pid = libc::pid_t::try_from(tid) - .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "TID does not fit pid_t"))?; - let remote_address = usize::try_from(address).map_err(|_| { - io::Error::new( - io::ErrorKind::InvalidInput, - "remote address does not fit usize", - ) - })?; - let local = libc::iovec { - iov_base: source.as_ptr().cast_mut().cast(), - iov_len: source.len(), - }; - let remote = libc::iovec { - iov_base: remote_address as *mut libc::c_void, - iov_len: source.len(), - }; - - // SAFETY: the local iovec spans the caller-provided live buffer. The - // remote address is untrusted but bounded; the kernel validates that it is - // writable in the target process. - let copied = retry_eintr(|| unsafe { - libc::process_vm_writev( - pid, - std::ptr::addr_of!(local), - 1, - std::ptr::addr_of!(remote), - 1, - 0, - ) - }); - match copied { - Ok(copied) => require_exact(copied, source.len(), "task-memory write"), - Err(error) if syscall_profile_denied(&error) => { - write_exact_to_proc_mem(tid, address, source) - .map_err(|fallback| fallback_error("write", &error, fallback)) - } - Err(error) => Err(error), - } -} - fn syscall_profile_denied(error: &io::Error) -> bool { matches!( error.raw_os_error(), @@ -145,22 +98,14 @@ fn read_exact_from_proc_mem(tid: u32, address: u64, destination: &mut [u8]) -> i require_exact(copied, destination.len(), "proc task-memory read") } -fn write_exact_to_proc_mem(tid: u32, address: u64, source: &[u8]) -> io::Result<()> { - let file = std::fs::OpenOptions::new() - .write(true) - .open(format!("/proc/{tid}/mem"))?; - let copied = file.write_at(source, address)?; - require_exact(copied, source.len(), "proc task-memory write") -} - -/// Prove same-UID parent-to-child read and write access under the active Yama, -/// LSM, and outer seccomp posture. +/// Prove same-UID parent-to-child read access under the active Yama, LSM, and +/// outer seccomp posture. The broker only reads workload memory; it never +/// writes it. /// /// Call this only from a single-threaded probe process. The child executes /// raw, allocation-free syscalls between `fork` and `_exit`. pub fn probe_child_access() -> io::Result<()> { const INITIAL: u64 = 0x1122_3344_5566_7788; - const REPLACEMENT: u64 = 0xaabb_ccdd_eeff_0011; // SAFETY: mmap creates one private anonymous page owned by this process. let mapping = unsafe { libc::mmap( @@ -216,7 +161,6 @@ pub fn probe_child_access() -> io::Result<()> { if libc::prctl(libc::PR_SET_DUMPABLE, 1, 0, 0, 0) < 0 || write_eventfd(ready.as_raw_fd()).is_err() || read_eventfd(proceed.as_raw_fd()).is_err() - || mapping.cast::().read() != REPLACEMENT { libc::_exit(1); } @@ -237,11 +181,6 @@ pub fn probe_child_access() -> io::Result<()> { "cross-child memory read returned wrong data", )); } - write_exact( - u32::try_from(child).map_err(|_| io::Error::other("child PID does not fit u32"))?, - mapping_address, - &REPLACEMENT.to_ne_bytes(), - )?; write_eventfd(proceed.as_raw_fd())?; let mut status = 0; // SAFETY: child is a live direct child and status points to storage. @@ -354,9 +293,8 @@ mod tests { use super::*; #[test] - fn reads_and_writes_exact_same_process_memory() { + fn reads_exact_same_process_memory() { let source = 0x1122_3344_5566_7788_u64; - let mut destination = 0_u64; let mut bytes = [0_u8; size_of::()]; read_exact( @@ -366,21 +304,11 @@ mod tests { ) .expect("read source"); assert_eq!(u64::from_ne_bytes(bytes), source); - - let replacement = 0xaabb_ccdd_eeff_0011_u64; - write_exact( - std::process::id(), - std::ptr::addr_of_mut!(destination) as u64, - &replacement.to_ne_bytes(), - ) - .expect("write destination"); - assert_eq!(destination, replacement); } #[test] - fn proc_mem_fallback_reads_and_writes_exact_memory() { + fn proc_mem_fallback_reads_exact_memory() { let source = 0x0102_0304_0506_0708_u64; - let mut destination = 0_u64; let mut bytes = [0_u8; size_of::()]; read_exact_from_proc_mem( std::process::id(), @@ -389,14 +317,6 @@ mod tests { ) .expect("read through proc mem"); assert_eq!(u64::from_ne_bytes(bytes), source); - - write_exact_to_proc_mem( - std::process::id(), - std::ptr::addr_of_mut!(destination) as u64, - &source.to_ne_bytes(), - ) - .expect("write through proc mem"); - assert_eq!(destination, source); } #[test] diff --git a/crates/openshell-sandbox-backend/src/boundary_protocol.rs b/crates/openshell-sandbox-backend/src/boundary_protocol.rs index 6c6c66e76c..0cf4efcec8 100644 --- a/crates/openshell-sandbox-backend/src/boundary_protocol.rs +++ b/crates/openshell-sandbox-backend/src/boundary_protocol.rs @@ -94,9 +94,6 @@ pub struct SeccompEvidence { pub retained_socket_operation: bool, pub proc_fd_identity: bool, pub task_memory_read: bool, - pub task_memory_write: bool, - pub cancellation: bool, - pub task_memory_writes_disabled: bool, } /// Mechanism-specific audit evidence for the native Linux sandbox adapter. @@ -124,6 +121,10 @@ pub struct NativeLinuxSandboxAuditEvidence { pub tcp_dns_round_trip: bool, pub tcp_allow_round_trip: bool, pub tcp_deny_round_trip: bool, + /// Workload INET sockets are bound to loopback before injection, the + /// binding cannot be changed from sandbox credentials, and accepted + /// sockets inherit it. Native local `accept` depends on this property. + pub socket_loopback_confinement: bool, } impl NativeLinuxSandboxAuditEvidence { @@ -143,14 +144,13 @@ impl NativeLinuxSandboxAuditEvidence { && self.seccomp.retained_socket_operation && self.seccomp.proc_fd_identity && self.seccomp.task_memory_read - && self.seccomp.task_memory_write - && (self.seccomp.cancellation || self.seccomp.task_memory_writes_disabled) && self.landlock_abi >= 3 && self.landlock_allow_deny && self.udp_dns_round_trip && self.tcp_dns_round_trip && self.tcp_allow_round_trip - && self.tcp_deny_round_trip; + && self.tcp_deny_round_trip + && self.socket_loopback_confinement; if complete { Ok(()) } else { @@ -175,14 +175,14 @@ impl NativeLinuxSandboxAuditEvidence { && self.udp_dns_round_trip && self.tcp_dns_round_trip && self.tcp_allow_round_trip - && self.tcp_deny_round_trip, + && self.tcp_deny_round_trip + && self.socket_loopback_confinement, "seccomp-notify", ), request_attribution: EnforcedProperty::new( self.seccomp.id_validation && self.seccomp.proc_fd_identity - && self.seccomp.task_memory_read - && self.seccomp.task_memory_write, + && self.seccomp.task_memory_read, "seccomp-notify-procfs", ), privilege_floor: EnforcedProperty::new( @@ -1455,9 +1455,6 @@ mod tests { retained_socket_operation: true, proc_fd_identity: true, task_memory_read: true, - task_memory_write: true, - cancellation: true, - task_memory_writes_disabled: false, }, landlock_abi: 6, landlock_allow_deny: true, @@ -1465,6 +1462,7 @@ mod tests { tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, } } @@ -1489,19 +1487,11 @@ mod tests { } #[test] - fn audit_evidence_accepts_legacy_read_only_listener() { + fn audit_evidence_requires_socket_loopback_confinement() { let mut audit = complete_audit_evidence(); - audit.seccomp.cancellation = false; - audit.seccomp.task_memory_writes_disabled = true; - assert!(audit.validate().is_ok()); - } - - #[test] - fn audit_evidence_rejects_plain_listener_with_writes_enabled() { - let mut audit = complete_audit_evidence(); - audit.seccomp.cancellation = false; - audit.seccomp.task_memory_writes_disabled = false; + audit.socket_loopback_confinement = false; assert!(audit.validate().is_err()); + assert!(!audit.properties().egress_interception.enforced); } #[test] diff --git a/crates/openshell-sandbox-backend/src/runtime.rs b/crates/openshell-sandbox-backend/src/runtime.rs index 641d50cd24..fea7c29d27 100644 --- a/crates/openshell-sandbox-backend/src/runtime.rs +++ b/crates/openshell-sandbox-backend/src/runtime.rs @@ -2906,9 +2906,6 @@ mod tests { retained_socket_operation: true, proc_fd_identity: true, task_memory_read: true, - task_memory_write: true, - cancellation: true, - task_memory_writes_disabled: false, }, landlock_abi: 3, landlock_allow_deny: true, @@ -2916,6 +2913,7 @@ mod tests { tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, }; openshell_isolation_interface::contract::BoundaryConfirmation { generation: "test-generation".to_string(), diff --git a/crates/openshell-sandbox/src/accept_interrupt.rs b/crates/openshell-sandbox/src/accept_interrupt.rs deleted file mode 100644 index 86bbf39218..0000000000 --- a/crates/openshell-sandbox/src/accept_interrupt.rs +++ /dev/null @@ -1,297 +0,0 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -//! Cancellation for broker-owned blocking accepts without changing workload OFDs. -//! -//! SIGUSR2 is reserved by the sandbox binary. Its process-global disposition is -//! necessarily kernel state, not a global application context. All registration, -//! cancellation and thread ownership state belongs to one broker instance. - -#![allow(unsafe_code)] - -use std::collections::HashMap; -use std::io; -use std::marker::PhantomData; -use std::rc::Rc; -use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, Condvar, Mutex}; -use std::time::Duration; - -const INTERRUPT_SIGNAL: libc::c_int = libc::SIGUSR2; -const INTERRUPT_INTERVAL: Duration = Duration::from_millis(10); - -extern "C" fn interrupt_accept(_: libc::c_int) {} - -fn reserve_signal() -> io::Result<()> { - // SAFETY: both actions are initialized storage. The no-op handler is - // async-signal-safe and deliberately omits SA_RESTART so accept returns EINTR. - unsafe { - let mut previous: libc::sigaction = std::mem::zeroed(); - if libc::sigaction(INTERRUPT_SIGNAL, std::ptr::null(), &raw mut previous) < 0 { - return Err(io::Error::last_os_error()); - } - if previous.sa_sigaction != libc::SIG_DFL - && previous.sa_sigaction != interrupt_accept as *const () as usize - { - return Err(io::Error::other( - "sandbox SIGUSR2 is already reserved by another handler", - )); - } - let mut action: libc::sigaction = std::mem::zeroed(); - action.sa_sigaction = interrupt_accept as *const () as usize; - libc::sigemptyset(&raw mut action.sa_mask); - if libc::sigaction(INTERRUPT_SIGNAL, &raw const action, std::ptr::null_mut()) < 0 { - return Err(io::Error::last_os_error()); - } - } - Ok(()) -} - -#[derive(Default)] -struct State { - workers: Mutex>, - changed: Condvar, - stopped: AtomicBool, -} - -// musl represents pthread_t as an opaque pointer, unlike glibc's integer. It -// is only passed back to pthread_kill, never dereferenced by this module. -struct RegisteredThread(libc::pthread_t); - -// SAFETY: POSIX permits signaling a live pthread from another thread. The -// handle is accessed only under State::workers, and the owning worker removes -// its registration under that same mutex before returning. AcceptRegistration -// cannot move to another thread, so its Drop cannot outlive the owning worker. -unsafe impl Send for RegisteredThread {} - -pub struct AcceptMonitor { - state: Arc, - thread: Option>, -} - -impl AcceptMonitor { - pub(crate) fn start(valid: impl Fn(u64) -> bool + Send + 'static) -> io::Result { - reserve_signal()?; - let state = Arc::new(State::default()); - let worker_state = state.clone(); - let thread = std::thread::Builder::new() - .name("openshell-accept-cancellation".into()) - .spawn(move || monitor(&worker_state, valid))?; - Ok(Self { - state, - thread: Some(thread), - }) - } - - pub(crate) fn registrar(&self) -> AcceptRegistrar { - AcceptRegistrar(self.state.clone()) - } -} - -impl Drop for AcceptMonitor { - fn drop(&mut self) { - let workers = lock(&self.state.workers); - self.state.stopped.store(true, Ordering::Release); - self.state.changed.notify_all(); - drop(workers); - // The monitor keeps interrupting registered workers during shutdown. - // Registrations are removed before their threads can exit/reuse IDs. - if let Some(thread) = self.thread.take() { - let _ = thread.join(); - } - } -} - -#[derive(Clone)] -pub struct AcceptRegistrar(Arc); - -impl AcceptRegistrar { - pub(crate) fn register(&self, notification_id: u64) -> io::Result { - // SAFETY: this changes only the current broker worker's signal mask. - // Workload launchers do not inherit this mask; exec resets the handler. - let thread = unsafe { - let mut mask: libc::sigset_t = std::mem::zeroed(); - libc::sigemptyset(&raw mut mask); - libc::sigaddset(&raw mut mask, INTERRUPT_SIGNAL); - let error = - libc::pthread_sigmask(libc::SIG_UNBLOCK, &raw const mask, std::ptr::null_mut()); - if error != 0 { - return Err(io::Error::from_raw_os_error(error)); - } - libc::pthread_self() - }; - let mut workers = lock(&self.0.workers); - if self.0.stopped.load(Ordering::Acquire) { - return Err(io::Error::from_raw_os_error(libc::ECANCELED)); - } - if workers.contains_key(¬ification_id) { - return Err(io::Error::other( - "duplicate accept notification registration", - )); - } - workers.insert(notification_id, RegisteredThread(thread)); - self.0.changed.notify_one(); - Ok(AcceptRegistration { - state: self.0.clone(), - notification_id, - owning_thread: PhantomData, - }) - } -} - -pub struct AcceptRegistration { - state: Arc, - notification_id: u64, - // Drop must run on the registering thread before its pthread_t can expire. - // No Rc is allocated; this marker makes the guard neither Send nor Sync. - owning_thread: PhantomData>, -} - -impl AcceptRegistration { - pub(crate) fn ensure_running(&self) -> io::Result<()> { - if self.state.stopped.load(Ordering::Acquire) { - Err(io::Error::from_raw_os_error(libc::ECANCELED)) - } else { - Ok(()) - } - } -} - -impl Drop for AcceptRegistration { - fn drop(&mut self) { - lock(&self.state.workers).remove(&self.notification_id); - self.state.changed.notify_one(); - } -} - -fn monitor(state: &State, valid: impl Fn(u64) -> bool) { - let mut workers = lock(&state.workers); - loop { - let stopped = state.stopped.load(Ordering::Acquire); - if stopped && workers.is_empty() { - return; - } - for (¬ification_id, thread) in &*workers { - if stopped || !valid(notification_id) { - // SAFETY: the registration lock pins this live pthread_t. - // Repeated interrupts close the check-to-accept race: a signal - // received before accept cannot leave a later accept stranded. - let _ = unsafe { libc::pthread_kill(thread.0, INTERRUPT_SIGNAL) }; - } - } - workers = if workers.is_empty() { - state - .changed - .wait(workers) - .unwrap_or_else(std::sync::PoisonError::into_inner) - } else { - state - .changed - .wait_timeout(workers, INTERRUPT_INTERVAL) - .unwrap_or_else(std::sync::PoisonError::into_inner) - .0 - }; - } -} - -fn lock(mutex: &Mutex) -> std::sync::MutexGuard<'_, T> { - mutex - .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) -} - -#[cfg(test)] -mod tests { - use super::*; - use std::net::{TcpListener, TcpStream}; - use std::os::fd::AsRawFd; - - #[test] - fn registrar_crosses_threads_but_registration_ends_before_worker_exit() { - fn assert_send_sync() {} - assert_send_sync::(); - - let monitor = AcceptMonitor::start(|_| true).unwrap(); - let registrar = monitor.registrar(); - std::thread::spawn(move || { - let registration = registrar.register(3).unwrap(); - assert!(registrar.register(3).is_err()); - assert!(lock(®istrar.0.workers).contains_key(&3)); - drop(registration); - assert!(lock(®istrar.0.workers).is_empty()); - }) - .join() - .unwrap(); - assert!(lock(&monitor.state.workers).is_empty()); - } - - #[test] - fn cancellation_interrupts_competing_accept_after_readiness_was_consumed() { - let valid = Arc::new(AtomicBool::new(true)); - let monitored = valid.clone(); - let monitor = AcceptMonitor::start(move |_| monitored.load(Ordering::Acquire)).unwrap(); - let listener = TcpListener::bind("127.0.0.1:0").unwrap(); - let client = TcpStream::connect(listener.local_addr().unwrap()).unwrap(); - // Both contenders could observe this same readable listener. Consume - // its only connection before the second contender actually accepts. - let accepted = listener.accept().unwrap(); - let registrar = monitor.registrar(); - let (ready_tx, ready_rx) = std::sync::mpsc::channel(); - let (done_tx, done_rx) = std::sync::mpsc::channel(); - let worker = std::thread::spawn(move || { - let registration = registrar.register(1).unwrap(); - ready_tx.send(()).unwrap(); - // SAFETY: the listener is live and null address outputs are valid. - // Use the syscall directly: std::net retries EINTR internally. - let result = unsafe { - libc::accept4( - listener.as_raw_fd(), - std::ptr::null_mut(), - std::ptr::null_mut(), - libc::SOCK_CLOEXEC, - ) - }; - assert_eq!(result, -1); - let error = io::Error::last_os_error(); - assert_eq!(error.kind(), io::ErrorKind::Interrupted); - drop(registration); - done_tx.send(()).unwrap(); - }); - ready_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - valid.store(false, Ordering::Release); - done_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - worker.join().unwrap(); - drop((accepted, client, monitor)); - } - - #[test] - fn shutdown_interrupts_registered_accepts_and_reclaims_the_monitor() { - let monitor = AcceptMonitor::start(|_| true).unwrap(); - let registrar = monitor.registrar(); - let listener = TcpListener::bind("127.0.0.1:0").unwrap(); - let (ready_tx, ready_rx) = std::sync::mpsc::channel(); - let (done_tx, done_rx) = std::sync::mpsc::channel(); - let worker = std::thread::spawn(move || { - let registration = registrar.register(2).unwrap(); - ready_tx.send(()).unwrap(); - // SAFETY: owned listener and optional null address outputs. - let result = unsafe { - libc::accept4( - listener.as_raw_fd(), - std::ptr::null_mut(), - std::ptr::null_mut(), - libc::SOCK_CLOEXEC, - ) - }; - assert_eq!(result, -1); - assert!(registration.ensure_running().is_err()); - drop(registration); - done_tx.send(()).unwrap(); - }); - ready_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - let shutdown = std::thread::spawn(move || drop(monitor)); - done_rx.recv_timeout(Duration::from_secs(2)).unwrap(); - worker.join().unwrap(); - shutdown.join().unwrap(); - } -} diff --git a/crates/openshell-sandbox/src/boundary_exec.rs b/crates/openshell-sandbox/src/boundary_exec.rs index da35294c06..5482629069 100644 --- a/crates/openshell-sandbox/src/boundary_exec.rs +++ b/crates/openshell-sandbox/src/boundary_exec.rs @@ -168,7 +168,7 @@ impl LocalBoundaryExec { command.env("SHELL", shell); } for (key, value) in &self.user_environment { - if !key.starts_with("OPENSHELL_") { + if !key.starts_with(crate::process::RESERVED_ENV_PREFIX) { command.env(key, value); } } @@ -184,7 +184,7 @@ impl LocalBoundaryExec { } crate::process::strip_proxy_env_std(&mut command); for (key, value) in &spec.env { - if !key.starts_with("OPENSHELL_") { + if !key.starts_with(crate::process::RESERVED_ENV_PREFIX) { command.env(key, value); } } diff --git a/crates/openshell-sandbox/src/boundary_io.rs b/crates/openshell-sandbox/src/boundary_io.rs index d404739af0..1889052421 100644 --- a/crates/openshell-sandbox/src/boundary_io.rs +++ b/crates/openshell-sandbox/src/boundary_io.rs @@ -220,6 +220,25 @@ impl BoundaryRuntimeState { .is_ok_and(|groups| !groups.is_empty()) } + /// Whether any workload process remains, registered or not. + /// + /// A registered root is unregistered once it is reaped, but descendants + /// that ignored `SIGTERM` may outlive it. When the sandbox owns the + /// process tree (PID 1 or a child subreaper), every live descendant is + /// counted so termination is not reported complete while one survives. + #[must_use] + pub fn has_owned_processes(&self) -> bool { + if self.has_registered_processes() { + return true; + } + // An unreadable /proc fails closed: processes may remain. + #[cfg(target_os = "linux")] + return owned_processes(&[], self.exclusive_pid_namespace) + .map_or(true, |owned| !owned.is_empty()); + #[cfg(not(target_os = "linux"))] + false + } + /// End the boundary because required standing enforcement was lost. /// /// Returns `true` only to the caller that won the active-to-terminated @@ -265,13 +284,11 @@ impl BoundaryRuntimeState { // requiring ptrace or a capability. let mut previous = Vec::new(); for _ in 0..4 { - let owned = owned_process_ids(&roots, self.exclusive_pid_namespace); - for pid in &owned { - if roots.contains(pid) { - continue; - } - if let Ok(pid) = i32::try_from(*pid) { - let _ = nix::sys::signal::kill(nix::unistd::Pid::from_raw(pid), signal); + let owned = + owned_processes(&roots, self.exclusive_pid_namespace).unwrap_or_default(); + for process in &owned { + if !roots.contains(&process.pid) { + signal_owned_process(*process, signal); } } if owned == previous { @@ -283,13 +300,54 @@ impl BoundaryRuntimeState { } } +/// One scanned workload process, identified by PID and kernel start time so a +/// reused PID is never mistaken for it. #[cfg(target_os = "linux")] -fn owned_process_ids(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { - let mut parents = HashMap::new(); - let Ok(entries) = std::fs::read_dir("/proc") else { - return roots.to_vec(); - }; - for entry in entries.flatten() { +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord)] +struct OwnedProcess { + pid: u32, + start_time: u64, +} + +#[cfg(target_os = "linux")] +struct ProcStat { + parent: u32, + start_time: u64, + live: bool, +} + +#[cfg(target_os = "linux")] +fn read_proc_stat(pid: u32) -> Option { + let stat = std::fs::read_to_string(format!("/proc/{pid}/stat")).ok()?; + // The command name may contain spaces or parentheses; fields resume after + // the final ") ". Field 3 is the state, 4 the parent, 22 the start time. + let fields = stat + .rsplit_once(") ")? + .1 + .split_whitespace() + .collect::>(); + Some(ProcStat { + parent: fields.get(1)?.parse().ok()?, + start_time: fields.get(19)?.parse().ok()?, + live: !matches!(*fields.first()?, "Z" | "X" | "x"), + }) +} + +/// Whether orphaned descendants are reparented to this sandbox process. +#[cfg(target_os = "linux")] +fn sandbox_owns_process_tree() -> bool { + std::process::id() == 1 + || rustix::process::child_subreaper().is_ok_and(|subreaper| subreaper.is_some()) +} + +#[cfg(target_os = "linux")] +fn owned_processes( + roots: &[u32], + exclusive_pid_namespace: bool, +) -> std::io::Result> { + let mut stats = HashMap::new(); + let mut children: HashMap> = HashMap::new(); + for entry in std::fs::read_dir("/proc")?.flatten() { let Some(pid) = entry .file_name() .to_str() @@ -297,52 +355,65 @@ fn owned_process_ids(roots: &[u32], exclusive_pid_namespace: bool) -> Vec { else { continue; }; - let Ok(stat) = std::fs::read_to_string(entry.path().join("stat")) else { - continue; - }; - let Some(after_name) = stat.rsplit_once(") ").map(|(_, fields)| fields) else { - continue; - }; - let Some(parent) = after_name - .split_whitespace() - .nth(1) - .and_then(|field| field.parse::().ok()) - else { + if let Some(stat) = read_proc_stat(pid) { + children.entry(stat.parent).or_default().push(pid); + stats.insert(pid, stat); + } + } + + // When the sandbox is PID 1 of its exclusive namespace or a child + // subreaper, orphans are reparented to it, so its descendants are exactly + // the workload tree. Otherwise walk only from registered roots so unit + // tests and development runs cannot affect sibling tasks. + let sandbox = std::process::id(); + let mut pending = if exclusive_pid_namespace && sandbox_owns_process_tree() { + vec![sandbox] + } else { + roots.to_vec() + }; + let mut visited = std::collections::HashSet::new(); + let mut owned = Vec::new(); + while let Some(pid) = pending.pop() { + if !visited.insert(pid) { continue; - }; - parents.insert(pid, parent); - } - - // When openshell-sandbox is PID 1, every other process in its exclusive - // namespace is workload-owned, including an orphan reparented during the - // scan. Outside that deployment shape, restrict the walk to registered - // roots so unit tests and development runs cannot affect sibling tasks. - if exclusive_pid_namespace && std::process::id() == 1 { - let mut owned = parents - .keys() - .copied() - .filter(|pid| *pid != 1) - .collect::>(); - owned.sort_unstable(); - return owned; - } - - let mut owned = roots.to_vec(); - loop { - let mut changed = false; - for (&pid, &parent) in &parents { - if !owned.contains(&pid) && owned.contains(&parent) { - owned.push(pid); - changed = true; - } } - if !changed { - break; + if let Some(descendants) = children.get(&pid) { + pending.extend(descendants); + } + if let Some(stat) = stats.get(&pid) + && pid != sandbox + && stat.live + { + owned.push(OwnedProcess { + pid, + start_time: stat.start_time, + }); } } owned.sort_unstable(); - owned.dedup(); - owned + Ok(owned) +} + +/// Signal one scanned process through a pidfd, after confirming the pidfd +/// refers to the scanned process rather than a later process with its PID. +#[cfg(target_os = "linux")] +fn signal_owned_process(process: OwnedProcess, signal: nix::sys::signal::Signal) { + let Some(pid) = i32::try_from(process.pid) + .ok() + .and_then(rustix::process::Pid::from_raw) + else { + return; + }; + let Some(signal) = rustix::process::Signal::from_named_raw(signal as i32) else { + return; + }; + let Ok(pidfd) = rustix::process::pidfd_open(pid, rustix::process::PidfdFlags::empty()) else { + return; + }; + if read_proc_stat(process.pid).map(|stat| stat.start_time) != Some(process.start_time) { + return; + } + let _ = rustix::process::pidfd_send_signal(&pidfd, signal); } #[derive(Clone)] @@ -509,6 +580,73 @@ mod tests { )); } + #[cfg(target_os = "linux")] + #[test] + fn termination_waits_for_descendants_that_outlive_their_root() { + use std::io::BufRead as _; + use std::os::unix::process::CommandExt as _; + + // Becoming a subreaper changes this whole process, so run the + // scenario in a fresh copy of the test binary. + const CHILD_MARKER: &str = "OPENSHELL_SUBREAPER_TEARDOWN_CHILD"; + if std::env::var_os(CHILD_MARKER).is_none() { + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "boundary_io::tests::termination_waits_for_descendants_that_outlive_their_root", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + .expect("run isolated teardown test"); + assert!(status.success(), "isolated teardown test failed"); + return; + } + rustix::process::set_child_subreaper(Some(rustix::process::getpid())) + .expect("become child subreaper"); + let runtime = BoundaryRuntimeState::new_exclusive_pid_namespace(); + // The grandchild inherits an ignored SIGTERM; the root restores the + // default disposition and exits on SIGTERM. + let mut root = std::process::Command::new("/bin/sh") + .args([ + "-c", + "trap '' TERM; sleep 600 & trap - TERM; echo ready; wait", + ]) + .process_group(0) + .stdout(std::process::Stdio::piped()) + .spawn() + .expect("spawn root"); + let mut line = String::new(); + std::io::BufReader::new(root.stdout.take().unwrap()) + .read_line(&mut line) + .unwrap(); + assert_eq!(line.trim(), "ready"); + let terminal = Arc::new(std::sync::atomic::AtomicBool::new(false)); + runtime + .register_process_group(root.id(), terminal.clone(), Arc::new(Mutex::new(()))) + .expect("register root"); + + assert!(runtime.begin_termination()); + root.wait().expect("root exits on SIGTERM"); + terminal.store(true, Ordering::Release); + runtime.unregister_process_group(root.id(), &terminal); + assert!(!runtime.has_registered_processes()); + assert!( + runtime.has_owned_processes(), + "a SIGTERM-ignoring grandchild must keep termination incomplete" + ); + + runtime.force_kill(); + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(5); + while runtime.has_owned_processes() { + assert!( + std::time::Instant::now() < deadline, + "forced termination left a descendant alive" + ); + std::thread::sleep(std::time::Duration::from_millis(20)); + } + } + #[test] fn freeze_blocks_new_operations_until_explicit_resume() { let runtime = BoundaryRuntimeState::new_exclusive_pid_namespace(); diff --git a/crates/openshell-sandbox/src/boundary_server.rs b/crates/openshell-sandbox/src/boundary_server.rs index f84f98d424..b1833eadda 100644 --- a/crates/openshell-sandbox/src/boundary_server.rs +++ b/crates/openshell-sandbox/src/boundary_server.rs @@ -85,16 +85,15 @@ mod linux { // NVML may traverse the persistenced socket directory during initialization; // WSL2 supplies GPU libraries under /usr/lib/wsl and the /dev/dxg device. const GPU_BASELINE_READ_ONLY: &[&str] = &["/run/nvidia-persistenced", "/usr/lib/wsl"]; - // CUDA opens device nodes read-write and writes thread names through - // /proc//task//comm during cuInit(). A /proc/self rule would bind - // to the launcher's inodes, not those of its workload children. + // CUDA opens device nodes read-write. Its thread-name writes through + // /proc//task//comm are served by open mediation, so /proc + // stays read-only. const GPU_BASELINE_READ_WRITE: &[&str] = &[ "/dev/nvidiactl", "/dev/nvidia-uvm", "/dev/nvidia-uvm-tools", "/dev/nvidia-modeset", "/dev/dxg", - "/proc", ]; fn duration_micros(duration: Duration) -> u64 { @@ -153,13 +152,7 @@ mod linux { continue; } if policy.filesystem.read_only.contains(&path) { - if path != Path::new("/proc") { - continue; - } - policy - .filesystem - .read_only - .retain(|allowed| allowed != &path); + continue; } policy.filesystem.read_write.push(path); modified = true; @@ -222,10 +215,15 @@ mod linux { } crate::sandbox::apply_supervisor_startup_hardening() .map_err(|error| format!("install sandbox process prelude: {error}"))?; - if nix::unistd::getpid().as_raw() == 1 { - crate::managed_children::start_orphan_reaper() - .map_err(|error| format!("start sandbox orphan reaper: {error}"))?; - } + // Keep orphaned workload descendants in this process tree so + // termination can find and kill them, then reap the adopted ones. + // PID 1 already receives orphans; elsewhere become a child subreaper. + if nix::unistd::getpid().as_raw() != 1 { + rustix::process::set_child_subreaper(Some(rustix::process::getpid())) + .map_err(|error| format!("become child subreaper: {error}"))?; + } + crate::managed_children::start_orphan_reaper() + .map_err(|error| format!("start sandbox orphan reaper: {error}"))?; let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() .map_err(|error| format!("start sandbox workload launcher: {error}"))?; let protected_control_port = match &config.listener { @@ -342,6 +340,16 @@ mod linux { .to_string(), ); } + // Workload sockets share the loopback interface with a loopback + // listener. + BoundaryListenerConfig::TlsTcp { address, .. } + if address.ip().to_canonical().is_loopback() => + { + return Err( + "boundary TLS listener must not bind a loopback address; workloads share the loopback interface" + .to_string(), + ); + } BoundaryListenerConfig::Vsock { control_port: 0, .. } => { @@ -1699,6 +1707,7 @@ mod linux { "frozen workload could not be resumed".to_string(), )); } + self.network_broker.set_workload_frozen(false); tracing::info!( connection_id = ?principal.connection_id(), "Sandbox Protocol connection recovered; workload resumed" @@ -1755,6 +1764,7 @@ mod linux { return; } if let Some(process) = &process { + self.network_broker.set_workload_frozen(true); let _ = process.boundary_runtime.freeze(); } *connection = SupervisorConnectionState::Frozen { recovery_id }; @@ -1881,12 +1891,12 @@ mod linux { async fn wait_for_process_tree_exit(process: &ManagedProcess, timeout: Duration) -> bool { let deadline = tokio::time::Instant::now() + timeout; - while process.boundary_runtime.has_registered_processes() + while process.boundary_runtime.has_owned_processes() && tokio::time::Instant::now() < deadline { tokio::time::sleep(Duration::from_millis(25)).await; } - !process.boundary_runtime.has_registered_processes() + !process.boundary_runtime.has_owned_processes() } fn shutdown(&self) { @@ -2406,6 +2416,7 @@ mod linux { tcp_dns_round_trip: self.qualification.tcp_dns_round_trip, tcp_allow_round_trip: self.qualification.tcp_allow_round_trip, tcp_deny_round_trip: self.qualification.tcp_deny_round_trip, + socket_loopback_confinement: self.qualification.socket_loopback_confinement, }; // The boundary reports mechanism evidence; the authenticated host // backend validates it before constructing a ConfirmedBoundary. @@ -3216,7 +3227,7 @@ mod linux { }) } BoundaryListenerConfig::TlsTcp { address, tls } => { - let listener = std::net::TcpListener::bind(address)?; + let listener = Self::bind_tcp(*address)?; listener.set_nonblocking(true)?; let server_config = Arc::new(load_tls_server_config(tls)?); Ok(Self::Tcp { @@ -3227,6 +3238,28 @@ mod linux { } } + /// Bind the TCP control listener, dropping loopback-interface ingress + /// before it listens so workload sockets cannot reach it through + /// loopback or the pod's own address. Configuration rejects loopback + /// addresses; tests bind them without the filter. + fn bind_tcp(address: std::net::SocketAddr) -> io::Result { + let socket = socket2::Socket::new( + socket2::Domain::for_address(address), + socket2::Type::STREAM, + Some(socket2::Protocol::TCP), + )?; + socket.set_cloexec(true)?; + socket.set_reuse_address(true)?; + if !address.ip().is_loopback() { + openshell_isolation_interface::linux::socket_confinement::reject_loopback_ingress( + &socket, + )?; + } + socket.bind(&address.into())?; + socket.listen(128)?; + Ok(socket.into()) + } + fn bind_vsock(port: u32) -> io::Result { let family = libc::sa_family_t::try_from(libc::AF_VSOCK).map_err(|_| { io::Error::new(io::ErrorKind::InvalidInput, "AF_VSOCK exceeds sa_family_t") @@ -4501,9 +4534,6 @@ mod linux { retained_socket_operation: true, proc_fd_identity: true, task_memory_read: true, - task_memory_write: true, - cancellation: true, - task_memory_writes_disabled: false, }, landlock_abi: 6, landlock_allow_deny: true, @@ -4511,6 +4541,7 @@ mod linux { tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, } } @@ -4645,6 +4676,46 @@ mod linux { validate_running_identity(&config.workload_identity, false).unwrap(); } + #[test] + fn tcp_control_listener_rejects_loopback_addresses() { + let directory = tempfile::tempdir().expect("temporary directory"); + let (server_tls, _client_tls) = stage_test_tls(directory.path(), "validate"); + let config = |address: &str| BoundaryConfig { + boundary_id: "sandbox-1".to_string(), + generation: "generation-1".to_string(), + session_id: test_session_id(), + session_rotation: openshell_core::jwt::SessionRotation::new(1) + .expect("session rotation"), + auth_epoch: CredentialEpoch::new(1).expect("auth epoch"), + gateway_id: "test-gateway".to_string(), + verification_keys: vec![test_verification_key()], + listener: BoundaryListenerConfig::TlsTcp { + address: address.parse().expect("valid address"), + tls: server_tls.clone(), + }, + resource_claims: std::collections::BTreeMap::new(), + resource_claim_files: std::collections::BTreeMap::new(), + workload_identity: test_workload_identity(), + outer_fence: test_outer_fence(), + child_env: std::collections::HashMap::new(), + }; + for address in [ + "127.0.0.1:5500", + "127.0.0.2:5500", + "[::1]:5500", + "[::ffff:127.0.0.1]:5500", + ] { + assert!( + validate_config(&config(address)).is_err(), + "{address} must be rejected" + ); + } + for address in ["0.0.0.0:5500", "[::]:5500", "10.42.0.7:5500"] { + validate_config(&config(address)) + .unwrap_or_else(|error| panic!("{address} must be accepted: {error}")); + } + } + #[test] fn runtime_resource_claim_file_must_match_admitted_claim() { let directory = tempfile::tempdir().expect("temporary directory"); @@ -4767,6 +4838,35 @@ mod linux { server.abort(); } + #[test] + fn pod_control_listener_rejects_loopback_ingress() { + let directory = tempfile::tempdir().expect("temporary directory"); + let (server_tls, _client_tls) = stage_test_tls(directory.path(), "loopback"); + let listener = ControlListener::bind(&BoundaryListenerConfig::TlsTcp { + address: "0.0.0.0:0".parse().expect("valid address"), + tls: server_tls, + }) + .expect("bind TLS listener"); + let port = listener + .tcp_local_addr() + .expect("TLS listener address") + .port(); + // Loopback and the host's own address both arrive on `lo`; the + // dropped SYN never completes a handshake. + let result = std::net::TcpStream::connect_timeout( + &std::net::SocketAddr::from(([127, 0, 0, 1], port)), + Duration::from_millis(300), + ); + assert!( + result.is_err(), + "loopback client reached the control listener" + ); + assert!(matches!( + listener.accept().map(|_| ()), + Err(error) if error.kind() == io::ErrorKind::WouldBlock + )); + } + #[test] fn tls_listener_preserves_session_when_control_switches_to_async_streaming() { let directory = tempfile::tempdir().expect("temporary directory"); diff --git a/crates/openshell-sandbox/src/lib.rs b/crates/openshell-sandbox/src/lib.rs index a8d31fbfe9..110a395842 100644 --- a/crates/openshell-sandbox/src/lib.rs +++ b/crates/openshell-sandbox/src/lib.rs @@ -3,8 +3,6 @@ //! Capability-free in-workload sandbox boundary. -#[cfg(target_os = "linux")] -mod accept_interrupt; pub mod boundary_exec; pub mod boundary_io; mod boundary_server; @@ -44,6 +42,7 @@ pub struct RuntimeQualification { pub tcp_dns_round_trip: bool, pub tcp_allow_round_trip: bool, pub tcp_deny_round_trip: bool, + pub socket_loopback_confinement: bool, } /// Placeholder used when compiling the package on a non-Linux host. diff --git a/crates/openshell-sandbox/src/main.rs b/crates/openshell-sandbox/src/main.rs index 99f37b202e..78f6fa514b 100644 --- a/crates/openshell-sandbox/src/main.rs +++ b/crates/openshell-sandbox/src/main.rs @@ -95,17 +95,14 @@ struct QualificationReport { task_memory_copy: bool, connected_send_fast_path: bool, socket_virtualization: bool, + /// Workload INET sockets are bound to loopback, the binding cannot be + /// changed from sandbox credentials, and accepted sockets inherit it. + socket_loopback_confinement: bool, dns_relay_bind: bool, udp_dns_round_trip: bool, tcp_dns_round_trip: bool, tcp_allow_round_trip: bool, tcp_deny_round_trip: bool, - wait_killable_recv: bool, - /// Selected seccomp listener cancellation mode: `killable` (>= 5.19) or - /// `legacy_read_only` (< 5.19, broker output writes disabled). - seccomp_listener_mode: &'static str, - /// Whether the broker disables task-memory output writes (legacy mode). - task_memory_writes_disabled: bool, } #[cfg(target_os = "linux")] @@ -163,6 +160,9 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi .into_diagnostic() .wrap_err("seccomp notification probe")?; probe_socket_virtualization().wrap_err("socket virtualization probe")?; + openshell_isolation_interface::linux::socket_confinement::probe_loopback_confinement() + .into_diagnostic() + .wrap_err("socket loopback confinement probe")?; probe_dns_relay_bind().wrap_err("DNS relay bind probe")?; let landlock_abi = openshell_isolation_interface::linux::landlock::abi_version() .into_diagnostic() @@ -196,18 +196,12 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi task_memory_copy, connected_send_fast_path: notification.connected_send_fast_path(), socket_virtualization: true, + socket_loopback_confinement: true, dns_relay_bind: true, udp_dns_round_trip: true, tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, - wait_killable_recv: notification.wait_killable_recv, - seccomp_listener_mode: if notification.wait_killable_recv { - "killable" - } else { - "legacy_read_only" - }, - task_memory_writes_disabled: !notification.wait_killable_recv, }; let qualification = openshell_sandbox::RuntimeQualification { seccomp: openshell_sandbox_backend::boundary_protocol::SeccompEvidence { @@ -218,11 +212,6 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi retained_socket_operation: true, proc_fd_identity: true, task_memory_read: task_memory_copy, - task_memory_write: task_memory_copy, - cancellation: notification.wait_killable_recv, - // Legacy plain listener (< 5.19) disables broker output writes; - // satisfies the `cancellation || writes_disabled` launch invariant. - task_memory_writes_disabled: !notification.wait_killable_recv, }, landlock_abi, landlock_allow_deny: true, @@ -230,6 +219,7 @@ fn qualify_runtime() -> Result<(openshell_sandbox::RuntimeQualification, Qualifi tcp_dns_round_trip: true, tcp_allow_round_trip: true, tcp_deny_round_trip: true, + socket_loopback_confinement: true, }; Ok((qualification, report)) } @@ -461,7 +451,6 @@ fn probe_socket_virtualization() -> Result<()> { openshell_isolation_interface::linux::seccomp_notify::install_listener(&[ libc::SYS_socket, libc::SYS_connect, - libc::SYS_getpeername, libc::SYS_sendto, ]); let Ok(listener) = listener else { @@ -516,7 +505,6 @@ fn probe_socket_virtualization() -> Result<()> { let mut observed_connect = false; let mut observed_dns_tcp_connect = false; let mut observed_denied_connect = false; - let mut observed_peer = false; let mut observed_dns_send = false; while !(observed_tcp_sockets == 3 @@ -524,7 +512,6 @@ fn probe_socket_virtualization() -> Result<()> { && observed_connect && observed_dns_tcp_connect && observed_denied_connect - && observed_peer && observed_dns_send) { let notification = listener @@ -650,27 +637,6 @@ fn probe_socket_virtualization() -> Result<()> { .respond_value(notification.id, 0) .into_diagnostic()?; } - libc::SYS_getpeername => { - let fd = i32::try_from(notification.args[0]) - .map_err(|_| miette::miette!("peer FD does not fit i32"))?; - let entry = registry.resolve(notification.tid, fd).into_diagnostic()?; - let SocketState::Connected { original_peer } = entry.state() else { - listener - .respond_errno(notification.id, libc::ENOTCONN) - .into_diagnostic()?; - return Err(miette::miette!("peer query preceded mediated connect")); - }; - write_probe_sockaddr( - notification.tid, - notification.args[1], - notification.args[2], - *original_peer, - )?; - listener - .respond_value(notification.id, 0) - .into_diagnostic()?; - observed_peer = true; - } libc::SYS_sendto => { let fd = i32::try_from(notification.args[0]) .map_err(|_| miette::miette!("sendto FD does not fit i32"))?; @@ -944,38 +910,6 @@ fn decode_probe_sockaddr(bytes: &[u8]) -> Result { ))) } -#[cfg(target_os = "linux")] -fn write_probe_sockaddr( - tid: u32, - address: u64, - length_address: u64, - peer: std::net::SocketAddr, -) -> Result<()> { - use std::mem::size_of; - - let (sockaddr, sockaddr_length) = encode_probe_sockaddr(peer)?; - let mut requested_length = [0_u8; size_of::()]; - openshell_isolation_interface::linux::task_memory::read_exact( - tid, - length_address, - &mut requested_length, - ) - .into_diagnostic()?; - let requested_length = libc::socklen_t::from_ne_bytes(requested_length); - if requested_length < sockaddr_length { - return Err(miette::miette!("peer sockaddr buffer is too small")); - } - openshell_isolation_interface::linux::task_memory::write_exact(tid, address, &sockaddr) - .into_diagnostic()?; - openshell_isolation_interface::linux::task_memory::write_exact( - tid, - length_address, - &sockaddr_length.to_ne_bytes(), - ) - .into_diagnostic()?; - Ok(()) -} - #[cfg(target_os = "linux")] #[allow(unsafe_code)] fn run_capability_socket_child(args: &[String]) -> Result<()> { diff --git a/crates/openshell-sandbox/src/network_broker.rs b/crates/openshell-sandbox/src/network_broker.rs index f2f196c925..8988ae3328 100644 --- a/crates/openshell-sandbox/src/network_broker.rs +++ b/crates/openshell-sandbox/src/network_broker.rs @@ -30,11 +30,9 @@ use tokio::sync::{mpsc, oneshot}; const SOCKET_CAPACITY: usize = 4_096; const SOCKET_FD_HEADROOM: usize = 64; const OPEN_QUEUE_CAPACITY: usize = 256; -const ACCEPT_WORKER_CAPACITY: usize = 64; const DNS_QUEUE_CAPACITY: usize = 256; const DNS_WORKER_CAPACITY: usize = 256; const DNS_QUERY_TIMEOUT: Duration = Duration::from_secs(10); -const ACCEPT_POLL_INTERVAL: Duration = Duration::from_millis(250); const DNS_RELAY_ADDRESS: SocketAddr = SocketAddr::V4(std::net::SocketAddrV4::new( Ipv4Addr::new(127, 0, 0, 53), 53, @@ -73,27 +71,6 @@ fn acquire_pending_dns_slot(active: &Arc) -> io::Result, -} - -impl Drop for PendingAcceptSlot { - fn drop(&mut self) { - self.active.fetch_sub(1, Ordering::AcqRel); - } -} - -fn acquire_pending_accept_slot(active: &Arc) -> io::Result { - active - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { - (current < ACCEPT_WORKER_CAPACITY).then_some(current + 1) - }) - .map_err(|_| io::Error::from_raw_os_error(libc::EAGAIN))?; - Ok(PendingAcceptSlot { - active: Arc::clone(active), - }) -} - fn acquire_pending_open_slot(active: &Arc) -> io::Result { active .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { @@ -191,13 +168,12 @@ fn register_dns_socket( #[derive(Clone)] struct NotificationQueues { provider_files: crate::provider_files::ProviderFiles, + workload_frozen: Arc, protected_control_port: Option, - accept_registrar: crate::accept_interrupt::AcceptRegistrar, identity_resolver: ProcfsIdentityResolver, pending: mpsc::Sender, dns_relay: DnsRelay, active_opens: Arc, - active_accepts: Arc, retained_socket_capacity: usize, decision_timeout: Duration, } @@ -206,7 +182,7 @@ struct NotificationQueues { #[derive(Clone)] pub struct NetworkBroker { provider_files: crate::provider_files::ProviderFiles, - _accept_monitor: Arc, + workload_frozen: Arc, pending: Arc>>, pending_dns: Arc>>, dns_address: SocketAddr, @@ -250,28 +226,23 @@ impl NetworkBroker { decision_timeout: Duration, ) -> io::Result { let listener = Arc::new(listener); - let monitor_listener = listener.clone(); - let accept_monitor = Arc::new(crate::accept_interrupt::AcceptMonitor::start(move |id| { - monitor_listener.validate_id(id).is_ok() - })?); let (pending_tx, pending_rx) = mpsc::channel(OPEN_QUEUE_CAPACITY); let (pending_dns_tx, pending_dns_rx) = mpsc::channel(DNS_QUEUE_CAPACITY); let active_opens = Arc::new(AtomicUsize::new(0)); - let active_accepts = Arc::new(AtomicUsize::new(0)); let dns_relay = start_dns_relay(dns_address, pending_dns_tx)?; let dns_address = dns_relay.address; let retained_socket_capacity = retained_socket_capacity()?; let registry = Arc::new(Mutex::new(SocketRegistry::new(1, SOCKET_CAPACITY)?)); let provider_files = crate::provider_files::ProviderFiles::default(); + let workload_frozen = Arc::new(AtomicBool::new(false)); let queues = NotificationQueues { provider_files: provider_files.clone(), + workload_frozen: workload_frozen.clone(), protected_control_port, - accept_registrar: accept_monitor.registrar(), identity_resolver: ProcfsIdentityResolver::for_pid_namespace(), pending: pending_tx, dns_relay, active_opens, - active_accepts, retained_socket_capacity, decision_timeout, }; @@ -295,28 +266,52 @@ impl NetworkBroker { break; } }; - if let Err(error) = dispatch_notification( - Arc::clone(®istry), - Arc::clone(&listener), - notification, - queues.clone(), - ) { - tracing::warn!( - tid = notification.tid, - syscall = notification.syscall, - %error, - "sandbox network notification denied (tid={}, syscall={}): {error}", - notification.tid, - notification.syscall - ); - let _ = listener.respond_errno(notification.id, error_to_errno(&error)); + // Contain a handler panic so one faulty notification + // cannot silently kill the broker and hang every blocked + // workload syscall. The failing syscall gets an error; the + // broker keeps mediating the rest. + let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + dispatch_notification( + Arc::clone(®istry), + Arc::clone(&listener), + notification, + queues.clone(), + ) + })); + match outcome { + Ok(Ok(())) => {} + Ok(Err(error)) => { + tracing::warn!( + tid = notification.tid, + syscall = notification.syscall, + %error, + "sandbox network notification denied (tid={}, syscall={}): {error}", + notification.tid, + notification.syscall + ); + let _ = + listener.respond_errno(notification.id, error_to_errno(&error)); + } + Err(_) => { + tracing::error!( + tid = notification.tid, + syscall = notification.syscall, + "sandbox network notification handler panicked (tid={}, syscall={})", + notification.tid, + notification.syscall + ); + let _ = listener.respond_errno(notification.id, libc::EIO); + } } } + // The broker thread is exiting; dependent operations must fail + // closed rather than block on a listener no one services. + broker_healthy.store(false, Ordering::Release); }) .map_err(|error| io::Error::other(format!("start network broker: {error}")))?; Ok(Self { provider_files, - _accept_monitor: accept_monitor, + workload_frozen, pending: Arc::new(tokio::sync::Mutex::new(pending_rx)), pending_dns: Arc::new(tokio::sync::Mutex::new(pending_dns_rx)), dns_address, @@ -324,6 +319,14 @@ impl NetworkBroker { }) } + /// Record whether the boundary has stopped the workload for supervisor + /// recovery. While frozen, workload requests to send `SIGCONT` are refused + /// so a process that was not yet stopped cannot resume the others. Set + /// this before stopping the workload and clear it after resuming it. + pub(crate) fn set_workload_frozen(&self, frozen: bool) { + self.workload_frozen.store(frozen, Ordering::Release); + } + pub(crate) async fn accept(&self) -> io::Result { self.pending .lock() @@ -549,13 +552,18 @@ fn dispatch_notification( &listener, notification, std::process::id(), + &queues.workload_frozen, ); } - if syscall == libc::SYS_tkill { + if matches!( + syscall, + libc::SYS_tkill | libc::SYS_tgkill | libc::SYS_rt_tgsigqueueinfo + ) { return openshell_isolation_interface::linux::process_signal::mediate_thread_signal( &listener, notification, std::process::id(), + &queues.workload_frozen, ); } if syscall == libc::SYS_socket { @@ -575,32 +583,18 @@ fn dispatch_notification( if syscall == libc::SYS_listen { return listen_socket(®istry, &listener, notification); } - if matches!(syscall, libc::SYS_accept | libc::SYS_accept4) { - return accept_socket( - registry, - listener, - notification, - queues.active_accepts, - queues.accept_registrar, - ); - } if matches!( syscall, libc::SYS_sendto | libc::SYS_sendmsg | libc::SYS_sendmmsg ) { return classify_send(®istry, &listener, notification, &queues.dns_relay); } - if syscall == libc::SYS_getpeername { - return get_peer_name(®istry, &listener, notification); - } if syscall == libc::SYS_setsockopt { let level = i32::try_from(notification.args[1]) .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; let option = i32::try_from(notification.args[2]) .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; - if (level == libc::IPPROTO_TCP && option == libc::TCP_FASTOPEN_CONNECT) - || (level == libc::IPPROTO_IPV6 && option == libc::IPV6_ADDRFORM) - { + if socket_option_is_denied(level, option) { return Err(io::Error::from_raw_os_error(libc::EPERM)); } return listener.respond_continue(notification.id); @@ -608,6 +602,32 @@ fn dispatch_notification( Err(io::Error::from_raw_os_error(libc::EPERM)) } +/// Options the workload may never set, decided from scalar syscall arguments +/// that another thread cannot replace before the kernel reads them. +/// +/// Interface-selection options could redirect or unpin a socket's loopback +/// device binding; the others enable Fast Open or change the address family. The kernel already refuses to change an existing binding +/// without `CAP_NET_RAW`; denying them here keeps confinement independent of +/// the capability state of the namespace that owns the network namespace. +fn socket_option_is_denied(level: i32, option: i32) -> bool { + matches!( + (level, option), + (libc::IPPROTO_TCP, libc::TCP_FASTOPEN_CONNECT) + | ( + libc::IPPROTO_IPV6, + libc::IPV6_ADDRFORM | libc::IPV6_UNICAST_IF | libc::IPV6_MULTICAST_IF + ) + | ( + libc::SOL_SOCKET, + libc::SO_BINDTODEVICE | libc::SO_BINDTOIFINDEX + ) + | ( + libc::IPPROTO_IP, + libc::IP_UNICAST_IF | libc::IP_MULTICAST_IF + ) + ) +} + fn create_socket( registry: &Mutex, listener: &NotificationListener, @@ -617,7 +637,13 @@ fn create_socket( let domain = i32::try_from(notification.args[0]) .map_err(|_| io::Error::from_raw_os_error(libc::EAFNOSUPPORT))?; if !matches!(domain, libc::AF_INET | libc::AF_INET6) { - return listener.respond_continue(notification.id); + // The workload filter already refuses other families. Repeat the + // decision here so a filter change cannot let a kernel transport + // socket bypass loopback confinement; the domain is a scalar argument. + if matches!(domain, libc::AF_UNIX | libc::AF_NETLINK) { + return listener.respond_continue(notification.id); + } + return Err(io::Error::from_raw_os_error(libc::EAFNOSUPPORT)); } let raw_kind = i32::try_from(notification.args[1]) .map_err(|_| io::Error::from_raw_os_error(libc::EPROTONOSUPPORT))?; @@ -654,6 +680,10 @@ fn create_socket( } // SAFETY: successful socket returned one owned descriptor. let source = unsafe { OwnedFd::from_raw_fd(source) }; + // Confinement is standing kernel state that must exist before the workload + // can observe the descriptor. Natively accepted children inherit it, so + // local accept needs no per-connection broker inspection. + openshell_isolation_interface::linux::socket_confinement::confine_to_loopback(&source)?; let metadata = SocketMetadata { family, kind, @@ -773,15 +803,21 @@ fn connect_socket( let destination = read_socket_addr(notification.tid, notification.args[1], notification.args[2])?; reject_protected_control_destination(destination, protected_control_port)?; - let (kind, socket_identity, nonblocking) = { + let (kind, socket_identity, nonblocking, repeated) = { let registry = lock(®istry); let entry = registry.resolve(notification.tid, fd)?; ( entry.metadata().kind, entry.identity(), entry.metadata().nonblocking, + repeated_connect_outcome(entry.state(), entry.metadata().kind, destination), ) }; + match repeated { + Some(0) => return listener.respond_value(notification.id, 0), + Some(errno) => return Err(io::Error::from_raw_os_error(errno)), + None => {} + } if kind == InetKind::DnsUdp && destination.port() == 0 { let mut registry = lock(®istry); let entry = registry.resolve_mut(notification.tid, fd)?; @@ -798,6 +834,7 @@ fn connect_socket( if entry.metadata().family != destination_family { return Err(io::Error::from_raw_os_error(libc::EAFNOSUPPORT)); } + listener.validate_id(notification.id)?; // glibc and uv use UDP connect(..., port 0), getsockname(), and an // AF_UNSPEC disconnect to rank resolved addresses. Bind only to the // matching loopback family and report success; never connect the @@ -820,6 +857,7 @@ fn connect_socket( return Err(io::Error::from_raw_os_error(libc::EISCONN)); } let source_fd = entry.retained_preconnect()?.as_raw_fd(); + listener.validate_id(notification.id)?; let peer = ensure_dns_source_bound(source_fd, entry.metadata().family)?; let admissions = match kind { InetKind::Tcp => &dns_relay.tcp_admissions, @@ -842,8 +880,21 @@ fn connect_socket( if destination.ip().is_loopback() && !openshell_core::google_cloud::is_metadata_destination(destination) { + if kind == InetKind::Tcp { + return connect_local_tcp( + ®istry, + &listener, + notification, + fd, + socket_identity, + destination, + &active_opens, + ); + } + // A UDP connect completes immediately. let mut registry = lock(®istry); let entry = registry.resolve_mut(notification.tid, fd)?; + listener.validate_id(notification.id)?; connect_exact(entry.retained_preconnect()?.as_raw_fd(), destination)?; entry.set_state(SocketState::Local { peer: destination }); entry.release_preconnect(); @@ -928,6 +979,136 @@ fn connect_socket( Ok(()) } +/// Connect a workload TCP socket to a loopback endpoint without blocking the +/// notification dispatcher. +/// +/// The broker connects a duplicate of its retained socket, which shares the +/// workload's open file, and never holds the registry lock while waiting. A +/// nonblocking socket gets the native `EINPROGRESS` and the kernel completes +/// the handshake on the shared socket; a blocking socket waits on a bounded +/// worker thread. A slow or full local listener therefore cannot stall +/// mediation of unrelated syscalls. A nonblocking socket is recorded as +/// connected once the handshake starts, so a repeated `connect` reports +/// `EISCONN` even while the handshake is still in progress. +fn connect_local_tcp( + registry: &Arc>, + listener: &Arc, + notification: Notification, + fd: RawFd, + socket_identity: SocketIdentity, + destination: SocketAddr, + active_opens: &Arc, +) -> io::Result<()> { + let connector = { + let registry = lock(registry); + let entry = registry.resolve(notification.tid, fd)?; + rustix::io::fcntl_dupfd_cloexec(entry.retained_preconnect()?, 3)? + }; + listener.validate_id(notification.id)?; + // SAFETY: F_GETFL reads the flags of the live shared open file. + let flags = unsafe { libc::fcntl(connector.as_raw_fd(), libc::F_GETFL) }; + if flags < 0 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK != 0 { + let started = with_sockaddr(destination, |pointer, length| { + // SAFETY: pointer/length describe a live sockaddr; the connector + // is a live duplicate of the workload's socket. + if unsafe { libc::connect(connector.as_raw_fd(), pointer, length) } == 0 { + Ok(()) + } else { + Err(io::Error::last_os_error()) + } + }); + let in_progress = started + .as_ref() + .is_err_and(|error| error.raw_os_error() == Some(libc::EINPROGRESS)); + if started.is_ok() || in_progress { + commit_local_connect(registry, notification.tid, fd, socket_identity, destination); + } + return match started { + Ok(()) => listener.respond_value(notification.id, 0), + Err(error) => Err(error), + }; + } + let slot = acquire_pending_open_slot(active_opens)?; + let registry = Arc::clone(registry); + let worker_listener = Arc::clone(listener); + std::thread::Builder::new() + .name("openshell-local-connect".to_string()) + .spawn(move || { + let _slot = slot; + // The duplicate shares the workload's blocking open file; wait + // natively without changing its flags. + let result = with_sockaddr(destination, |pointer, length| { + // SAFETY: pointer/length describe a live sockaddr and the + // connector is a live duplicate of the workload's socket. + if unsafe { libc::connect(connector.as_raw_fd(), pointer, length) } == 0 { + Ok(()) + } else { + Err(io::Error::last_os_error()) + } + }); + if result.is_ok() { + commit_local_connect( + ®istry, + notification.tid, + fd, + socket_identity, + destination, + ); + } + let _ = match result { + Ok(()) => worker_listener.respond_value(notification.id, 0), + Err(error) => { + worker_listener.respond_errno(notification.id, error_to_errno(&error)) + } + }; + }) + .map_err(|error| io::Error::other(format!("start local-connect worker: {error}")))?; + Ok(()) +} + +/// Record a completed or in-progress loopback TCP connect, unless the +/// descriptor now names a different socket. +fn commit_local_connect( + registry: &Mutex, + tid: u32, + fd: RawFd, + socket_identity: SocketIdentity, + destination: SocketAddr, +) { + let mut registry = lock(registry); + if let Ok(entry) = registry.resolve_mut(tid, fd) + && entry.identity() == socket_identity + { + entry.set_state(SocketState::Local { peer: destination }); + entry.release_preconnect(); + } +} + +/// Result for a `connect` on a socket the broker already connected, as the +/// kernel would report it: `Some(0)` for success, `Some(errno)` for an error, +/// `None` when the socket is not yet connected. +/// +/// A notified syscall interrupted by a signal is restarted after the broker +/// may already have completed it, so a repeat must not depend on the broker's +/// released pre-connect descriptor. +fn repeated_connect_outcome( + state: &SocketState, + kind: InetKind, + destination: SocketAddr, +) -> Option { + match state { + SocketState::Created | SocketState::Bound { .. } | SocketState::Listening { .. } => None, + // UDP connect replaces the association; repeating the same one succeeds. + SocketState::DnsUdp { relay } if *relay == destination => Some(0), + SocketState::Local { peer } if kind == InetKind::DnsUdp && *peer == destination => Some(0), + _ if kind == InetKind::Tcp => Some(libc::EISCONN), + _ => None, + } +} + const fn tcp_denial_errno(reason: TcpOpenDenial) -> i32 { match reason { TcpOpenDenial::PolicyDenied @@ -1037,6 +1218,12 @@ fn bind_socket( let bind_result = { let mut registry = lock(registry); let entry = registry.resolve_mut(notification.tid, fd)?; + // A native bind is never restarted, but a notified one can be after a + // signal. Report a repeat of the bind the broker completed as success. + if entry.state() == &(SocketState::Bound { local }) { + return listener.respond_value(notification.id, 0); + } + listener.validate_id(notification.id)?; bind_exact(entry.retained_preconnect()?.as_raw_fd(), local) }; if bind_result @@ -1082,6 +1269,8 @@ fn listen_socket( let Ok(entry) = registry.resolve_mut(notification.tid, fd) else { return listener.respond_continue(notification.id); }; + // listen(2) may be repeated natively, so a restart needs no special case. + listener.validate_id(notification.id)?; // SAFETY: retained descriptor is the exact registered socket OFD. if unsafe { libc::listen(entry.retained_preconnect()?.as_raw_fd(), backlog) } < 0 { return Err(io::Error::last_os_error()); @@ -1091,218 +1280,6 @@ fn listen_socket( listener.respond_value(notification.id, 0) } -fn accept_socket( - registry: Arc>, - listener: Arc, - notification: Notification, - active_accepts: Arc, - accept_registrar: crate::accept_interrupt::AcceptRegistrar, -) -> io::Result<()> { - let fd = raw_fd(notification.args[0])?; - let flags = if i64::from(notification.syscall) == libc::SYS_accept4 { - i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))? - } else { - 0 - }; - if flags & !(libc::SOCK_CLOEXEC | libc::SOCK_NONBLOCK) != 0 { - return Err(io::Error::from_raw_os_error(libc::EINVAL)); - } - if (notification.args[1] == 0) != (notification.args[2] == 0) { - return Err(io::Error::from_raw_os_error(libc::EFAULT)); - } - let (listener_inode, metadata, source) = { - let registry = lock(®istry); - let Ok(entry) = registry.resolve(notification.tid, fd) else { - return listener.respond_continue(notification.id); - }; - if !matches!(entry.state(), SocketState::Listening { .. }) - || entry.metadata().kind != InetKind::Tcp - { - return Err(io::Error::from_raw_os_error(libc::EINVAL)); - } - let source = duplicate_close_on_exec(entry.retained_preconnect()?.as_raw_fd())?; - (entry.identity().inode, entry.metadata(), source) - }; - let slot = acquire_pending_accept_slot(&active_accepts)?; - let worker_listener = Arc::clone(&listener); - std::thread::Builder::new() - .name("openshell-local-accept".to_string()) - .spawn(move || { - let _slot = slot; - let registration = match accept_registrar.register(notification.id) { - Ok(registration) => registration, - Err(error) => { - let _ = worker_listener.respond_errno(notification.id, error_to_errno(&error)); - return; - } - }; - if let Err(error) = accept_and_inject( - ®istry, - &worker_listener, - notification, - AcceptOperation { - flags, - listener_inode, - metadata, - source, - registration, - }, - ) { - let _ = worker_listener.respond_errno(notification.id, error_to_errno(&error)); - } - }) - .map_err(|error| io::Error::other(format!("start local-accept worker: {error}")))?; - Ok(()) -} - -struct AcceptOperation { - flags: i32, - listener_inode: u64, - metadata: SocketMetadata, - source: OwnedFd, - registration: crate::accept_interrupt::AcceptRegistration, -} - -fn accept_and_inject( - registry: &Mutex, - listener: &NotificationListener, - notification: Notification, - operation: AcceptOperation, -) -> io::Result<()> { - let AcceptOperation { - flags, - listener_inode, - metadata, - source, - registration, - } = operation; - let mut poll = libc::pollfd { - fd: source.as_raw_fd(), - events: libc::POLLIN, - revents: 0, - }; - // SAFETY: F_GETFL reads the live listener OFD flags. - let current_flags = unsafe { libc::fcntl(source.as_raw_fd(), libc::F_GETFL) }; - if current_flags < 0 { - return Err(io::Error::last_os_error()); - } - let nonblocking = current_flags & libc::O_NONBLOCK != 0; - let timeout = if nonblocking { - 0 - } else { - i32::try_from(ACCEPT_POLL_INTERVAL.as_millis()).map_err(io::Error::other)? - }; - // Readiness may disappear before accept (another accept or an aborted - // connection). The registered watchdog interrupts a blocked syscall when - // its notification dies or the broker shuts down. No workload OFD flags - // are changed, and no worker can outlive its cancellation registration. - loop { - registration.ensure_running()?; - listener.validate_id(notification.id)?; - // SAFETY: poll references one live pollfd for this call. - let ready = unsafe { libc::poll(&raw mut poll, 1, timeout) }; - if ready < 0 { - let error = io::Error::last_os_error(); - if error.kind() == io::ErrorKind::Interrupted { - continue; - } - return Err(error); - } - if ready == 0 { - if nonblocking { - return Err(io::Error::from_raw_os_error(libc::EAGAIN)); - } - continue; - } - break; - } - - let mut storage = std::mem::MaybeUninit::::zeroed(); - let mut length = - libc::socklen_t::try_from(size_of::()).map_err(io::Error::other)?; - // Always keep the broker-side descriptor close-on-exec. ADDFD separately - // applies the workload's requested descriptor flag. - let accepted_flags = flags | libc::SOCK_CLOEXEC; - // SAFETY: storage and length are live outputs and source is a listening - // socket proven by the registry. - let accepted = unsafe { - libc::accept4( - source.as_raw_fd(), - storage.as_mut_ptr().cast(), - &raw mut length, - accepted_flags, - ) - }; - if accepted < 0 { - return Err(io::Error::last_os_error()); - } - // Only the blocking accept phase needs asynchronous interruption. Stop - // monitoring before ADDFD completes the notification, otherwise a normal - // successful response could be mistaken for cancellation during commit. - drop(registration); - // SAFETY: successful accept4 returned one newly owned descriptor. - let accepted = unsafe { OwnedFd::from_raw_fd(accepted) }; - // SAFETY: accept4 initialized the reported prefix of storage. - let peer = decode_sockaddr( - unsafe { storage.assume_init() }, - usize::try_from(length).unwrap_or(0), - )?; - if !peer.ip().is_loopback() { - return Err(io::Error::from_raw_os_error(libc::EACCES)); - } - if notification.args[1] != 0 { - write_socket_addr( - listener, - notification.id, - notification.tid, - notification.args[1], - notification.args[2], - peer, - )?; - } - - let accepted_metadata = SocketMetadata { - family: metadata.family, - kind: InetKind::Tcp, - close_on_exec: flags & libc::SOCK_CLOEXEC != 0, - nonblocking: flags & libc::SOCK_NONBLOCK != 0, - creator_generation: u64::from(notification.tid), - }; - let mut registry = lock(registry); - let notifying_fd = raw_fd(notification.args[0])?; - if registry - .resolve(notification.tid, notifying_fd)? - .identity() - .inode - != listener_inode - { - return Err(io::Error::from_raw_os_error(libc::EBADF)); - } - if registry.is_full() { - collect_closed_socket_entries_locked(&mut registry)?; - } - let tentative = registry.stage(accepted, accepted_metadata)?; - listener.add_fd_and_send( - notification.id, - tentative.source_fd(), - accepted_metadata.close_on_exec, - )?; - registry.commit_with_state(tentative, SocketState::AcceptedLocal { peer })?; - Ok(()) -} - -fn duplicate_close_on_exec(fd: RawFd) -> io::Result { - // SAFETY: F_DUPFD_CLOEXEC returns an independent owned descriptor for the - // same open-file description. - let duplicate = unsafe { libc::fcntl(fd, libc::F_DUPFD_CLOEXEC, 3) }; - if duplicate < 0 { - return Err(io::Error::last_os_error()); - } - // SAFETY: successful fcntl returned one newly owned descriptor. - Ok(unsafe { OwnedFd::from_raw_fd(duplicate) }) -} - fn classify_send( registry: &Mutex, listener: &NotificationListener, @@ -1311,6 +1288,12 @@ fn classify_send( ) -> io::Result<()> { let fd = raw_fd(notification.args[0])?; let syscall = i64::from(notification.syscall); + // Fast Open turns a send into a connect. Decide from the scalar flags + // argument, which another thread cannot replace, so the denial also + // covers natively accepted and other unregistered descriptors. + if send_flags(syscall, notification.args) & libc::MSG_FASTOPEN != 0 { + return Err(io::Error::from_raw_os_error(libc::EPERM)); + } let (state, metadata) = { let registry = lock(registry); let Ok(entry) = registry.resolve(notification.tid, fd) else { @@ -1320,10 +1303,8 @@ fn classify_send( }; (entry.state().clone(), entry.metadata()) }; - if matches!( - &state, - SocketState::Connected { .. } | SocketState::AcceptedLocal { .. } - ) || (metadata.kind == InetKind::Tcp && matches!(&state, SocketState::Local { .. })) + if matches!(&state, SocketState::Connected { .. }) + || (metadata.kind == InetKind::Tcp && matches!(&state, SocketState::Local { .. })) { return listener.respond_continue(notification.id); } @@ -1332,9 +1313,6 @@ fn classify_send( libc::SYS_sendmsg => vec![read_sendmsg_message( notification.tid, notification.args[1], - i32::try_from(notification.args[2]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, - None, )?], libc::SYS_sendmmsg => read_sendmmsg_messages(notification)?, _ => return Err(io::Error::from_raw_os_error(libc::ENOSYS)), @@ -1388,58 +1366,52 @@ fn classify_send( { let entry = registry.resolve_mut(notification.tid, fd)?; let source_fd = entry.retained_preconnect()?.as_raw_fd(); + listener.validate_id(notification.id)?; let peer = ensure_dns_source_bound(source_fd, entry.metadata().family)?; register_dns_socket(&dns_relay.udp_admissions, peer, entry.identity())?; if let Err(error) = connect_exact(source_fd, dns_relay.address) { lock(&dns_relay.udp_admissions).remove(&peer); return Err(error); } - for message in &messages { - send_dns_message(source_fd, message)?; - if let Some(length_address) = message.result_length_address { - let length = u32::try_from(message.data.len()) - .map_err(|_| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - listener.write_task_output( - notification.id, - notification.tid, - length_address, - &length.to_ne_bytes(), - )?; - } - } + // The socket is pinned to the relay and bound to loopback. The + // kernel performs the send; ancillary data was rejected at read + // time and a loopback destination contains any per-message + // routing override that races the check. entry.set_state(SocketState::DnsUdp { relay: dns_relay.address, }); entry.release_preconnect(); - let result = if syscall == libc::SYS_sendmmsg { - i64::try_from(messages.len()).unwrap_or(i64::MAX) - } else { - i64::try_from(messages[0].data.len()).unwrap_or(i64::MAX) - }; - listener.respond_value(notification.id, result) + listener.respond_continue(notification.id) } Ok(_) => Err(io::Error::from_raw_os_error(libc::EDESTADDRREQ)), - // Non-INET sockets and accepted local sockets were never registered. - // The mandatory outer fence still prevents an external kernel route. + // Non-INET sockets and natively accepted sockets were never + // registered. Accepted sockets inherit their listener's loopback + // binding, and the mandatory outer fence remains an independent + // backstop against an external kernel route. Err(_) => listener.respond_continue(notification.id), } } +fn send_flags(syscall: i64, args: [u64; 6]) -> i32 { + let flags = match syscall { + libc::SYS_sendmsg => args[2], + libc::SYS_sendto | libc::SYS_sendmmsg => args[3], + _ => 0, + }; + // Syscall flag arguments are C ints; the kernel ignores the upper word. + #[allow( + clippy::cast_possible_truncation, + reason = "the kernel reads only the low 32 bits of the flags argument" + )] + let flags = flags as u32; + flags.cast_signed() +} + struct SendMessage { - data: Vec, destination: Option, - flags: i32, - result_length_address: Option, } fn read_sendto_message(notification: Notification) -> io::Result { - let length = usize::try_from(notification.args[2]) - .map_err(|_| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - if u16::try_from(length).is_err() { - return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); - } - let mut data = vec![0_u8; length]; - task_memory::read_exact(notification.tid, notification.args[1], &mut data)?; let destination = if notification.args[4] == 0 { None } else { @@ -1449,22 +1421,15 @@ fn read_sendto_message(notification: Notification) -> io::Result { notification.args[5], )?) }; - Ok(SendMessage { - data, - destination, - flags: i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?, - result_length_address: None, - }) + Ok(SendMessage { destination }) } -fn read_sendmsg_message( - tid: u32, - address: u64, - flags: i32, - result_length_address: Option, -) -> io::Result { +fn read_sendmsg_message(tid: u32, address: u64) -> io::Result { let header = read_task_value::(tid, address)?; + // Ancillary data can carry a per-message routing override (IP_PKTINFO). + // Refuse it rather than continue a send the broker did not inspect; a + // loopback destination additionally contains an override that races this + // check. if header.msg_controllen != 0 { return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); } @@ -1477,39 +1442,7 @@ fn read_sendmsg_message( u64::from(header.msg_namelen), )?) }; - #[cfg(target_env = "musl")] - let iov_count = usize::try_from(header.msg_iovlen) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; - #[cfg(not(target_env = "musl"))] - let iov_count = header.msg_iovlen; - if iov_count > 32 { - return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); - } - let mut data = Vec::new(); - for index in 0..iov_count { - let offset = index - .checked_mul(size_of::()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?; - let iov = read_task_value::( - tid, - (header.msg_iov as u64) - .checked_add(u64::try_from(offset).unwrap_or(u64::MAX)) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EOVERFLOW))?, - )?; - let start = data.len(); - let end = start - .checked_add(iov.iov_len) - .filter(|length| u16::try_from(*length).is_ok()) - .ok_or_else(|| io::Error::from_raw_os_error(libc::EMSGSIZE))?; - data.resize(end, 0); - task_memory::read_exact(tid, iov.iov_base as u64, &mut data[start..end])?; - } - Ok(SendMessage { - data, - destination, - flags, - result_length_address, - }) + Ok(SendMessage { destination }) } fn read_sendmmsg_messages(notification: Notification) -> io::Result> { @@ -1518,8 +1451,6 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result 32 { return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); } - let flags = i32::try_from(notification.args[3]) - .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; (0..count) .map(|index| { let offset = index @@ -1528,18 +1459,7 @@ fn read_sendmmsg_messages(notification: Notification) -> io::Result(tid: u32, address: u64) -> io::Result { Ok(unsafe { std::ptr::read_unaligned(bytes.as_ptr().cast::()) }) } -fn send_dns_message(fd: RawFd, message: &SendMessage) -> io::Result<()> { - // SAFETY: `fd` is the retained exact UDP socket and the buffer remains - // valid for the duration of the syscall. - let sent = unsafe { - libc::send( - fd, - message.data.as_ptr().cast(), - message.data.len(), - message.flags, - ) - }; - if sent < 0 { - return Err(io::Error::last_os_error()); - } - if usize::try_from(sent).ok() == Some(message.data.len()) { - Ok(()) - } else { - Err(io::Error::from_raw_os_error(libc::EIO)) - } -} - -fn get_peer_name( - registry: &Mutex, - listener: &NotificationListener, - notification: Notification, -) -> io::Result<()> { - let fd = raw_fd(notification.args[0])?; - let registry = lock(registry); - let Ok(entry) = registry.resolve(notification.tid, fd) else { - return listener.respond_continue(notification.id); - }; - let peer = match entry.state() { - SocketState::Connected { original_peer } => *original_peer, - SocketState::Local { peer } | SocketState::AcceptedLocal { peer } => *peer, - _ => return Err(io::Error::from_raw_os_error(libc::ENOTCONN)), - }; - write_socket_addr( - listener, - notification.id, - notification.tid, - notification.args[1], - notification.args[2], - peer, - )?; - listener.respond_value(notification.id, 0) -} - fn connect_exact(fd: RawFd, address: SocketAddr) -> io::Result<()> { // Never let a blocking connect pin the single notification dispatcher. // O_NONBLOCK is an OFD flag, so restore the workload's original setting @@ -1745,50 +1618,6 @@ fn decode_sockaddr(storage: libc::sockaddr_storage, length: usize) -> io::Result } } -fn write_socket_addr( - listener: &NotificationListener, - notification_id: u64, - tid: u32, - address: u64, - length_address: u64, - value: SocketAddr, -) -> io::Result<()> { - // A LegacyReadOnly listener (kernels < 5.19) cannot safely write into - // workload memory: without WAIT_KILLABLE_RECV the notified accept/ - // getpeername could resume and repurpose these buffers between validation - // and the broker write. Fail closed before reading or writing anything, so - // this address-writing path is inert in legacy mode. Callers that pass a - // null address argument (accept with a null peer address) never reach here. - if listener.writes_disabled() { - return Err(io::Error::from_raw_os_error(libc::EOPNOTSUPP)); - } - let mut supplied_length = [0_u8; size_of::()]; - task_memory::read_exact(tid, length_address, &mut supplied_length)?; - let supplied_length = libc::socklen_t::from_ne_bytes(supplied_length); - let (bytes, actual_length) = sockaddr_bytes(value)?; - let copied = usize::try_from(supplied_length) - .unwrap_or(0) - .min(bytes.len()); - if copied != 0 { - listener.write_task_output(notification_id, tid, address, &bytes[..copied])?; - } - listener.write_task_output( - notification_id, - tid, - length_address, - &actual_length.to_ne_bytes(), - ) -} - -fn sockaddr_bytes(address: SocketAddr) -> io::Result<(Vec, libc::socklen_t)> { - with_sockaddr(address, |native, length| { - let length_usize = usize::try_from(length).map_err(io::Error::other)?; - // SAFETY: with_sockaddr lends fully initialized storage for this call. - let bytes = unsafe { std::slice::from_raw_parts(native.cast::(), length_usize) }; - Ok((bytes.to_vec(), length)) - }) -} - fn with_sockaddr( address: SocketAddr, operation: impl FnOnce(*const libc::sockaddr, libc::socklen_t) -> io::Result, @@ -1846,7 +1675,7 @@ fn error_to_errno(error: &io::Error) -> i32 { #[cfg(test)] mod tests { use super::*; - use openshell_isolation_interface::linux::seccomp_notify::ListenerMode; + use openshell_isolation_interface::linux::socket_confinement; #[test] fn provider_files_are_opened_on_demand_and_replaced() { @@ -2026,25 +1855,6 @@ mod tests { ))); } - #[test] - fn legacy_listener_rejects_socket_addr_write() { - // accept-with-address and getpeername both route through - // write_socket_addr; on a LegacyReadOnly listener the path must fail - // closed (EOPNOTSUPP) before any task-memory access. - // SAFETY: dup returns a new descriptor or a negative error. - let dup = unsafe { libc::dup(libc::STDERR_FILENO) }; - assert!(dup >= 0, "dup stderr"); - let listener = NotificationListener::from_fd_with_mode( - // SAFETY: successful dup returned a new owned descriptor. - unsafe { OwnedFd::from_raw_fd(dup) }, - ListenerMode::LegacyReadOnly, - ); - let peer: SocketAddr = "127.0.0.1:8080".parse().unwrap(); - let error = write_socket_addr(&listener, 1, 0, 0, 0, peer) - .expect_err("legacy listener must reject socket-address writes"); - assert_eq!(error.raw_os_error(), Some(libc::EOPNOTSUPP)); - } - #[test] fn relay_rejects_descriptor_replaced_after_policy_decision() { let metadata = SocketMetadata { @@ -2056,11 +1866,10 @@ mod tests { }; let mut registry = SocketRegistry::new(1, 2).unwrap(); let mut create = || { - // SAFETY: a successful socket call returns a new owned descriptor. - let fd = unsafe { libc::socket(libc::AF_INET, libc::SOCK_STREAM, 0) }; - assert!(fd >= 0); - let socket = unsafe { OwnedFd::from_raw_fd(fd) }; - let installed = duplicate_close_on_exec(fd).unwrap(); + let socket = OwnedFd::from( + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None).unwrap(), + ); + let installed = rustix::io::fcntl_dupfd_cloexec(&socket, 3).unwrap(); let tentative = registry.stage(socket, metadata).unwrap(); let identity = registry.commit(tentative).unwrap(); (installed, identity) @@ -2383,42 +2192,39 @@ mod tests { } #[test] - fn accepted_loopback_stream_is_registered_for_notified_operations() { + fn native_accept_inherits_loopback_confinement() { let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() .expect("start workload launcher"); let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); let (ready_tx, ready_rx) = std::sync::mpsc::sync_channel(1); let workload = std::thread::spawn(move || { launcher - .execute(move || -> io::Result { - let listener = TcpListener::bind("127.0.0.1:0")?; - ready_tx - .send(listener.local_addr()?) - .map_err(|_| io::Error::other("test client disappeared"))?; - let (stream, _) = listener.accept()?; - let peer = stream.peer_addr()?; - let payload = b"accepted"; - let iov = libc::iovec { - iov_base: payload.as_ptr().cast_mut().cast(), - iov_len: payload.len(), - }; - let message = libc::msghdr { - msg_name: std::ptr::null_mut(), - msg_namelen: 0, - msg_iov: (&raw const iov).cast_mut(), - msg_iovlen: 1, - msg_control: std::ptr::null_mut(), - msg_controllen: 0, - msg_flags: 0, - }; - // SAFETY: message references one live immutable payload; - // the accepted stream remains open for the call. - let sent = unsafe { libc::sendmsg(stream.as_raw_fd(), &raw const message, 0) }; - if sent != isize::try_from(payload.len()).expect("payload fits isize") { - return Err(io::Error::last_os_error()); - } - Ok(peer) - }) + .execute( + move || -> io::Result<(SocketAddr, SocketAddr, Option>)> { + let listener = TcpListener::bind("127.0.0.1:0")?; + ready_tx + .send(listener.local_addr()?) + .map_err(|_| io::Error::other("test client disappeared"))?; + // std passes a peer-address buffer; native accept + // fills it directly from the kernel. + let (stream, accepted_peer) = listener.accept()?; + let peer = stream.peer_addr()?; + let device = socket_confinement::bound_device(&stream)?; + // sendmsg with no destination is notified and must + // continue for an untracked accepted stream. + let payload = b"accepted"; + let sent = rustix::net::sendmsg( + &stream, + &[io::IoSlice::new(payload)], + &mut rustix::net::SendAncillaryBuffer::default(), + rustix::net::SendFlags::empty(), + )?; + if sent != payload.len() { + return Err(io::Error::from_raw_os_error(libc::EIO)); + } + Ok((accepted_peer, peer, device)) + }, + ) .expect("launcher result") }); @@ -2435,14 +2241,360 @@ mod tests { .read_exact(&mut payload) .expect("read accepted stream"); assert_eq!(&payload, b"accepted"); + let (accepted_peer, peer, device) = workload + .join() + .expect("join workload") + .expect("accepted workload"); + let client_address = client.local_addr().unwrap(); + assert_eq!(accepted_peer, client_address); + assert_eq!(peer, client_address); + assert_eq!(device.as_deref(), Some(&b"lo"[..])); + } + + /// Bound device name and the errno from an attempted rebind. + type ConfinementObservation = (Option>, Option); + + #[test] + fn workload_sockets_are_bound_to_loopback_and_cannot_be_rebound() { + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let results = launcher + .execute(|| -> io::Result> { + let mut results = Vec::new(); + for (domain, kind) in [ + (socket2::Domain::IPV4, socket2::Type::STREAM), + (socket2::Domain::IPV4, socket2::Type::DGRAM), + (socket2::Domain::IPV6, socket2::Type::STREAM), + ] { + let socket = socket2::Socket::new(domain, kind, None)?; + let device = socket_confinement::bound_device(&socket)?; + let rebind_error = socket + .bind_device(Some(b"eth0")) + .err() + .and_then(|error| error.raw_os_error()); + results.push((device, rebind_error)); + } + Ok(results) + }) + .expect("launcher result") + .expect("workload sockets"); + for (device, rebind_error) in results { + assert_eq!(device.as_deref(), Some(&b"lo"[..])); + assert_eq!(rebind_error, Some(libc::EPERM)); + } + } + + #[test] + fn fast_open_sends_are_denied_for_every_descriptor() { + let local = TcpListener::bind("127.0.0.1:0").unwrap(); + local.set_nonblocking(true).unwrap(); + let address = local.local_addr().unwrap(); + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let error = launcher + .execute(move || -> io::Result<()> { + let socket = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None)?; + socket + .send_to_with_flags(b"x", &address.into(), libc::MSG_FASTOPEN) + .map(drop) + }) + .unwrap() + .unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EPERM)); + assert_eq!( + local.accept().unwrap_err().kind(), + io::ErrorKind::WouldBlock + ); + } + + #[test] + fn send_flags_read_the_scalar_argument_for_each_syscall() { + let flags = u64::try_from(libc::MSG_FASTOPEN).unwrap(); + assert_eq!( + send_flags(libc::SYS_sendto, [0, 0, 0, flags, 0, 0]), + libc::MSG_FASTOPEN + ); + assert_eq!( + send_flags(libc::SYS_sendmsg, [0, 0, flags, 0, 0, 0]), + libc::MSG_FASTOPEN + ); + assert_eq!( + send_flags(libc::SYS_sendmmsg, [0, 0, 0, flags | (1 << 32), 0, 0]), + libc::MSG_FASTOPEN + ); + } + + #[test] + fn broker_refuses_socket_families_it_cannot_confine() { + // Independent of the static workload filter: the broker continues + // only Unix and netlink sockets and creates INET sockets itself. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let results = launcher + .execute(|| { + [ + (libc::AF_UNIX, socket2::Type::STREAM), + (libc::AF_RXRPC, socket2::Type::DGRAM), + (libc::AF_ALG, socket2::Type::SEQPACKET), + ] + .map(|(domain, kind)| { + socket2::Socket::new(socket2::Domain::from(domain), kind, None) + .map(drop) + .map_err(|error| error.raw_os_error()) + }) + }) + .expect("launcher result"); + assert_eq!( + results, + [ + Ok(()), + Err(Some(libc::EAFNOSUPPORT)), + Err(Some(libc::EAFNOSUPPORT)) + ] + ); + } + + #[test] + fn repeated_connects_report_what_the_kernel_would() { + let relay: SocketAddr = "127.0.0.53:53".parse().unwrap(); + let peer: SocketAddr = "127.0.0.1:8080".parse().unwrap(); + let other: SocketAddr = "127.0.0.1:9090".parse().unwrap(); + for (state, kind, destination, expected) in [ + (SocketState::Created, InetKind::Tcp, peer, None), + ( + SocketState::Bound { local: peer }, + InetKind::Tcp, + peer, + None, + ), + ( + SocketState::Local { peer }, + InetKind::Tcp, + peer, + Some(libc::EISCONN), + ), + ( + SocketState::Connected { + original_peer: "203.0.113.7:443".parse().unwrap(), + }, + InetKind::Tcp, + peer, + Some(libc::EISCONN), + ), + ( + SocketState::DnsTcp { relay }, + InetKind::Tcp, + relay, + Some(libc::EISCONN), + ), + ( + SocketState::DnsUdp { relay }, + InetKind::DnsUdp, + relay, + Some(0), + ), + (SocketState::DnsUdp { relay }, InetKind::DnsUdp, other, None), + (SocketState::Local { peer }, InetKind::DnsUdp, peer, Some(0)), + ] { + assert_eq!( + repeated_connect_outcome(&state, kind, destination), + expected, + "{state:?} {kind:?} {destination}" + ); + } + } + + #[test] + fn repeated_bind_and_connect_after_completion() { + // A signal can restart a syscall the broker already completed. A + // repeated connect reports EISCONN; a repeated bind of the same + // address succeeds, unlike a native EINVAL, so a restart is safe. + let service = TcpListener::bind("127.0.0.1:0").unwrap(); + let address = service.local_addr().unwrap(); + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let (bind_again, connect_again) = launcher + .execute(move || -> io::Result<(io::Result<()>, Option)> { + let socket = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None)?; + let local: SocketAddr = "127.0.0.1:0".parse().unwrap(); + socket.bind(&local.into())?; + let bind_again = socket.bind(&local.into()); + socket.connect(&address.into())?; + let connect_again = socket + .connect(&address.into()) + .err() + .and_then(|error| error.raw_os_error()); + Ok((bind_again, connect_again)) + }) + .expect("launcher result") + .expect("workload socket"); + bind_again.expect("repeated bind of the same address"); + assert_eq!(connect_again, Some(libc::EISCONN)); + } + + #[test] + fn frozen_workload_cannot_resume_processes_with_sigcont() { + // A workload process that is not yet stopped when the boundary + // freezes must not resume the others, through process-directed + // (kill) or thread-directed (tgkill) signals. + const CHILD_MARKER: &str = "OPENSHELL_FROZEN_SIGCONT_CHILD"; + if std::env::var_os(CHILD_MARKER).is_some() { + let errno = |result: nix::Result<()>| result.err().map_or(0, |error| error as i32); + let kill = errno(nix::sys::signal::kill( + nix::unistd::getpid(), + nix::sys::signal::Signal::SIGCONT, + )); + // SAFETY: tgkill takes scalar arguments naming this thread. + let tgkill = unsafe { + libc::syscall( + libc::SYS_tgkill, + libc::getpid(), + libc::syscall(libc::SYS_gettid), + libc::SIGCONT, + ) + }; + let tgkill = if tgkill < 0 { + io::Error::last_os_error().raw_os_error().unwrap_or(-1) + } else { + 0 + }; + println!("kill={kill} tgkill={tgkill}"); + return; + } + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let run_workload = |launcher: &openshell_isolation_interface::linux::workload_launcher::WorkloadLauncher| { + let output = launcher + .execute(|| { + std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "network_broker::tests::frozen_workload_cannot_resume_processes_with_sigcont", + "--nocapture", + "--quiet", + ]) + .env(CHILD_MARKER, "1") + .output() + }) + .unwrap() + .expect("run workload child"); + String::from_utf8_lossy(&output.stdout) + .lines() + .find(|line| line.starts_with("kill=")) + .expect("workload child result") + .to_string() + }; + broker.set_workload_frozen(true); + assert_eq!( + run_workload(&launcher), + format!("kill={} tgkill={}", libc::EPERM, libc::EPERM) + ); + broker.set_workload_frozen(false); + assert_eq!(run_workload(&launcher), "kill=0 tgkill=0"); + } + + #[test] + fn slow_loopback_connect_does_not_stall_other_mediation() { + // A listener that never accepts, with a full backlog, makes further + // connects wait. Other mediated syscalls must not wait behind them, + // whether the pending connect is blocking or nonblocking. + let saturated = + socket2::Socket::new(socket2::Domain::IPV4, socket2::Type::STREAM, None).unwrap(); + saturated + .bind(&"127.0.0.1:0".parse::().unwrap().into()) + .unwrap(); + saturated.listen(0).unwrap(); + let address = saturated.local_addr().unwrap().as_socket().unwrap(); + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let (blocking_wait, nonblocking_result) = launcher + .execute(move || { + // Fill the accept queue; later connects to it stall. + let filled = TcpStream::connect(address).expect("fill accept queue"); + let pending = std::thread::spawn(move || TcpStream::connect(address)); + std::thread::sleep(Duration::from_millis(500)); + let started = Instant::now(); + drop(UdpSocket::bind("127.0.0.1:0")); + let blocking_wait = started.elapsed(); + // A nonblocking connect reports progress immediately. + let nonblocking = socket2::Socket::new( + socket2::Domain::IPV4, + socket2::Type::STREAM.nonblocking(), + None, + ) + .unwrap(); + let nonblocking_result = nonblocking + .connect(&address.into()) + .err() + .and_then(|error| error.raw_os_error()); + // Leave the pending connect running; closing the listener when + // the test ends resets it. + drop(pending); + drop(filled); + (blocking_wait, nonblocking_result) + }) + .expect("launcher result"); assert!( - workload - .join() - .expect("join workload") - .expect("accepted workload") - .ip() - .is_loopback() + blocking_wait < Duration::from_secs(2), + "mediated socket creation waited {blocking_wait:?} behind a slow connect" ); + assert_eq!(nonblocking_result, Some(libc::EINPROGRESS)); + } + + #[test] + fn workload_cannot_bind_a_non_loopback_source_address() { + // Workload sockets can only present loopback source addresses. A + // non-loopback peer on a loopback-bound workload listener is therefore + // a non-workload process in the same network namespace. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let _broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let errors = launcher + .execute(|| { + ["192.0.2.10:0", "[2001:db8::10]:0"].map(|address| { + UdpSocket::bind(address) + .err() + .and_then(|error| error.raw_os_error()) + }) + }) + .expect("launcher result"); + assert_eq!(errors, [Some(libc::EACCES), Some(libc::EACCES)]); + } + + #[test] + fn interface_selection_options_are_denied() { + for (level, option) in [ + (libc::SOL_SOCKET, libc::SO_BINDTODEVICE), + (libc::SOL_SOCKET, libc::SO_BINDTOIFINDEX), + (libc::IPPROTO_IP, libc::IP_UNICAST_IF), + (libc::IPPROTO_IP, libc::IP_MULTICAST_IF), + (libc::IPPROTO_IPV6, libc::IPV6_UNICAST_IF), + (libc::IPPROTO_IPV6, libc::IPV6_MULTICAST_IF), + (libc::IPPROTO_IPV6, libc::IPV6_ADDRFORM), + (libc::IPPROTO_TCP, libc::TCP_FASTOPEN_CONNECT), + ] { + assert!(socket_option_is_denied(level, option), "{level}/{option}"); + } + assert!(!socket_option_is_denied( + libc::SOL_SOCKET, + libc::SO_REUSEADDR + )); + assert!(!socket_option_is_denied( + libc::IPPROTO_TCP, + libc::TCP_NODELAY + )); + assert!(!socket_option_is_denied( + libc::IPPROTO_IPV6, + libc::IPV6_V6ONLY + )); } #[test] @@ -2554,6 +2706,110 @@ mod tests { ); } + #[test] + fn udp_dns_after_connect_sends_with_sendmmsg() { + // glibc connects the resolver socket to the nameserver, then sends A + // and AAAA together with sendmmsg and no destination. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let dns_address = broker.dns_address(); + let client = std::thread::spawn(move || { + launcher + .execute(move || -> io::Result>> { + let socket = UdpSocket::bind("0.0.0.0:0")?; + socket.set_read_timeout(Some(Duration::from_secs(5)))?; + socket.connect(dns_address)?; + let queries = [&b"dns-query-a"[..], &b"dns-query-aaaa"[..]]; + let iovecs = queries.map(|query| [io::IoSlice::new(query)]); + let mut controls = [ + rustix::net::SendAncillaryBuffer::default(), + rustix::net::SendAncillaryBuffer::default(), + ]; + let [first, second] = &mut controls; + let mut messages = [ + rustix::net::MMsgHdr::new(&iovecs[0], first), + rustix::net::MMsgHdr::new(&iovecs[1], second), + ]; + let sent = rustix::net::sendmmsg( + &socket, + &mut messages, + rustix::net::SendFlags::empty(), + )?; + if sent != 2 { + return Err(io::Error::other("sendmmsg sent too few")); + } + let mut responses = Vec::new(); + for _ in 0..2 { + let mut response = [0_u8; 32]; + let length = socket.recv(&mut response)?; + responses.push(response[..length].to_vec()); + } + responses.sort(); + Ok(responses) + }) + .expect("launcher result") + }); + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("test runtime"); + for _ in 0..2 { + let query = runtime.block_on(broker.accept_dns()).expect("DNS query"); + let response = if query.request == b"dns-query-a" { + b"dns-response-a".to_vec() + } else if query.request == b"dns-query-aaaa" { + b"dns-response-aaaa".to_vec() + } else { + panic!("unexpected DNS query: {:?}", query.request); + }; + query.complete(Ok(response)).unwrap(); + } + assert_eq!( + client.join().expect("join client").expect("DNS client"), + vec![b"dns-response-a".to_vec(), b"dns-response-aaaa".to_vec()] + ); + } + + #[test] + fn dns_send_with_ancillary_data_is_refused() { + // Ancillary data can carry a per-message routing override such as + // IP_PKTINFO. The kernel performs mediated DNS sends, so control data + // is refused. + let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() + .expect("start workload launcher"); + let broker = NetworkBroker::start_for_test(listener).expect("start network broker"); + let dns_address = broker.dns_address(); + let errno = launcher + .execute(move || { + let socket = UdpSocket::bind("0.0.0.0:0").unwrap(); + let native = socket2::SockAddr::from(dns_address); + let payload = *b"dns-query"; + let mut iov = libc::iovec { + iov_base: payload.as_ptr().cast_mut().cast(), + iov_len: payload.len(), + }; + // Any non-empty control buffer is refused. + let mut control = [0_u8; 32]; + let header = libc::msghdr { + msg_name: native.as_ptr().cast_mut().cast(), + msg_namelen: native.len(), + msg_iov: &raw mut iov, + msg_iovlen: 1, + msg_control: control.as_mut_ptr().cast(), + msg_controllen: control.len(), + msg_flags: 0, + }; + // SAFETY: the header references live local buffers for the call. + let sent = unsafe { libc::sendmsg(socket.as_raw_fd(), &raw const header, 0) }; + (sent < 0) + .then(|| io::Error::last_os_error().raw_os_error()) + .flatten() + }) + .expect("launcher result"); + assert_eq!(errno, Some(libc::EOPNOTSUPP)); + } + #[test] fn udp_dns_allows_repeated_destination_sends_to_the_pinned_relay() { let (launcher, listener) = openshell_isolation_interface::linux::workload_launcher::start() diff --git a/crates/openshell-sandbox/src/process.rs b/crates/openshell-sandbox/src/process.rs index 595ceac29e..20812ce315 100644 --- a/crates/openshell-sandbox/src/process.rs +++ b/crates/openshell-sandbox/src/process.rs @@ -105,6 +105,9 @@ pub(crate) fn ca_runtime_read_only_paths(ca_paths: Option<&(PathBuf, PathBuf)>) paths } +/// Prefix of environment variable names reserved for `OpenShell`. +pub(crate) const RESERVED_ENV_PREFIX: &str = "OPENSHELL_"; + const SUPERVISOR_ONLY_ENV_VARS: &[&str] = &[ openshell_core::sandbox_env::OCI_IMAGE_USER, openshell_core::sandbox_env::SANDBOX_UID, @@ -205,6 +208,21 @@ fn apply_canonical_process_environment( interactive: bool, user_environment: &HashMap, ) { + // The canonical process inherits the sandbox's environment so the image's + // own ENV (PATH, LANG, JAVA_HOME, ...) reaches the workload. Remove the + // reserved OPENSHELL_ namespace inherited from the sandbox itself, which + // carries its own control state (for example the serialized user + // environment and log level), then restore the one marker the workload is + // meant to see. The gateway rejects declared variables in this namespace. + for (key, _) in std::env::vars_os() { + if key + .to_str() + .is_some_and(|key| key.starts_with(RESERVED_ENV_PREFIX)) + { + cmd.env_remove(key); + } + } + cmd.env(openshell_core::sandbox_env::SANDBOX, "1"); cmd.envs(user_environment); let (session_user, session_home) = session_user_and_home(policy, workspace.home()); // Resolve a shell present in the workload image. This code runs inside the @@ -450,9 +468,7 @@ impl ProcessHandle { provider_env: &HashMap, ) -> Result { let mut cmd = Command::new(program); - cmd.args(args) - .kill_on_drop(true) - .env(openshell_core::sandbox_env::SANDBOX, "1"); + cmd.args(args).kill_on_drop(true); let mut pty_master = None; let mut terminal_slave_fd = None; @@ -615,9 +631,7 @@ impl ProcessHandle { provider_env: &HashMap, ) -> Result { let mut cmd = Command::new(program); - cmd.args(args) - .kill_on_drop(true) - .env(openshell_core::sandbox_env::SANDBOX, "1"); + cmd.args(args).kill_on_drop(true); let mut pty_master = None; let mut terminal_slave_fd = None; @@ -1122,6 +1136,73 @@ mod tests { assert_eq!(variables.get("TERM"), Some(&"xterm-256color")); } + #[cfg(unix)] + #[test] + fn canonical_process_drops_inherited_reserved_environment() { + // The sandbox's own control variables live in the reserved + // OPENSHELL_ namespace and must not reach the workload, while the + // image's ordinary ENV must. Run in a fresh copy of the test binary + // so the test harness environment is untouched. + const CHILD_MARKER: &str = "OPENSHELL_TEST_RESERVED_ENV_CHILD"; + if std::env::var_os(CHILD_MARKER).is_none() { + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "process::tests::canonical_process_drops_inherited_reserved_environment", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .env(openshell_core::sandbox_env::LOG_LEVEL, "debug") + .env(openshell_core::sandbox_env::USER_ENVIRONMENT, "{}") + .env("IMAGE_LANG", "keep") + .status() + .expect("run isolated environment test"); + assert!(status.success(), "isolated environment test failed"); + return; + } + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .unwrap(); + let current_user = User::from_uid(nix::unistd::geteuid()).unwrap().unwrap(); + let policy = policy_with_process(ProcessPolicy { + run_as_user: Some(current_user.name), + run_as_group: None, + }); + // Mirror production: inherit the sandbox environment, no env_clear. + let mut cmd = Command::new("/usr/bin/env"); + cmd.stdout(StdStdio::piped()); + apply_canonical_process_environment( + &mut cmd, + &policy, + &ResolvedWorkspace::default(), + false, + &HashMap::from([("DECLARED".into(), "yes".into())]), + ); + let output = runtime + .block_on(async { cmd.output().await }) + .expect("run environment probe"); + assert!(output.status.success()); + let environment = String::from_utf8(output.stdout).unwrap(); + let variables: HashMap<_, _> = environment + .lines() + .filter_map(|line| line.split_once('=')) + .collect(); + assert!( + !variables + .keys() + .any(|key| key.starts_with(RESERVED_ENV_PREFIX) + && *key != openshell_core::sandbox_env::SANDBOX), + "reserved variables reached the workload: {variables:?}" + ); + assert_eq!( + variables.get(openshell_core::sandbox_env::SANDBOX), + Some(&"1") + ); + assert_eq!(variables.get("IMAGE_LANG"), Some(&"keep")); + assert_eq!(variables.get("DECLARED"), Some(&"yes")); + } + #[cfg(unix)] #[tokio::test] async fn canonical_process_receives_declared_environment_and_home() { diff --git a/crates/openshell-sandbox/src/provider_files.rs b/crates/openshell-sandbox/src/provider_files.rs index ab719cd355..384eb6b31f 100644 --- a/crates/openshell-sandbox/src/provider_files.rs +++ b/crates/openshell-sandbox/src/provider_files.rs @@ -17,6 +17,7 @@ use openshell_isolation_interface::linux::seccomp_notify::{Notification, Notific use openshell_isolation_interface::linux::task_memory; const PREFIX: &str = "/run/openshell/providers/"; +const PROC_PREFIX: &str = "/proc/"; const MAX_FILE_BYTES: usize = 65_536; const MAX_TOTAL_BYTES: usize = 262_144; const MAX_PATH_BYTES: usize = 4_096; @@ -75,6 +76,9 @@ impl ProviderFiles { } else { notification.args[0] }; + if handle_thread_comm_open(listener, notification, path_address)? { + return Ok(()); + } // Every workload open reaches the listener. Copy only the reserved // prefix for ordinary paths; full path reads are rare. let mut prefix = [0_u8; PREFIX.len()]; @@ -160,6 +164,97 @@ fn validate_path(path: &str) -> io::Result<()> { Ok(()) } +/// Serve a write open of the caller's own thread name file. +/// +/// `pthread_setname_np` and CUDA's `cuInit` rename threads by writing +/// `/proc//task//comm`. Landlock keeps `/proc` read-only, so the +/// broker opens the caller's own `comm` file and injects the descriptor; no +/// syscall is continued. The kernel's `comm_write` accepts a write only from +/// the target's own thread group, so a substituted path or reused thread ID +/// cannot rename another process's thread through the descriptor. Returns +/// `false` when the open is not such a request and normal mediation applies. +fn handle_thread_comm_open( + listener: &NotificationListener, + notification: Notification, + path_address: u64, +) -> io::Result { + let Ok(flags) = open_flags(¬ification) else { + return Ok(false); + }; + let access = flags & libc::O_ACCMODE; + // Reads are already allowed by the read-only /proc rule. + if access == libc::O_RDONLY { + return Ok(false); + } + let mut prefix = [0_u8; PROC_PREFIX.len()]; + if task_memory::read_exact(notification.tid, path_address, &mut prefix).is_err() + || prefix != PROC_PREFIX.as_bytes() + { + return Ok(false); + } + let Ok(path) = read_path(notification.tid, path_address) else { + return Ok(false); + }; + let Some(caller_group) = thread_group_of(notification.tid) else { + return Ok(false); + }; + let Some(target) = comm_target(&path, notification.tid, caller_group) else { + return Ok(false); + }; + // Anything else, including another process's thread, is left to + // Landlock, which denies the write. + if thread_group_of(target) != Some(caller_group) { + return Ok(false); + } + // A shell redirect opens with O_CREAT|O_TRUNC; both are no-ops on an + // existing comm file. O_EXCL fails as it would natively. + if flags & libc::O_EXCL != 0 { + listener.respond_errno(notification.id, libc::EEXIST)?; + return Ok(true); + } + if flags & (libc::O_TMPFILE | libc::O_DIRECTORY | libc::O_PATH) != 0 { + listener.respond_errno(notification.id, libc::EINVAL)?; + return Ok(true); + } + let file = std::fs::OpenOptions::new() + .read(access == libc::O_RDWR) + .write(true) + .open(format!("/proc/{caller_group}/task/{target}/comm"))?; + listener.add_fd_and_send( + notification.id, + file.as_raw_fd(), + flags & libc::O_CLOEXEC != 0, + )?; + Ok(true) +} + +/// Resolve the thread whose `comm` file `path` names, if it is the caller's +/// own thread group. +fn comm_target(path: &str, caller_tid: u32, caller_group: u32) -> Option { + let parts = path + .strip_prefix(PROC_PREFIX)? + .split('/') + .collect::>(); + let own_group = |part: &str| part == "self" || part.parse::().ok() == Some(caller_group); + match parts.as_slice() { + ["thread-self", "comm"] => Some(caller_tid), + [group, "comm"] if own_group(group) => Some(caller_group), + [group, "task", tid, "comm"] if own_group(group) => tid.parse().ok(), + _ => None, + } +} + +/// Thread group (process) ID of a thread, from its procfs status. +fn thread_group_of(tid: u32) -> Option { + std::fs::read_to_string(format!("/proc/{tid}/status")) + .ok()? + .lines() + .find_map(|line| line.strip_prefix("Tgid:"))? + .trim() + .parse() + .ok() +} + fn read_path(tid: u32, mut address: u64) -> io::Result { if address == 0 { return Err(io::Error::from_raw_os_error(libc::EFAULT)); @@ -228,12 +323,32 @@ fn sealed_memfd(content: &[u8]) -> io::Result { #[cfg(test)] mod tests { - use super::{ProviderFiles, sealed_memfd}; + use super::{ProviderFiles, comm_target, sealed_memfd}; use std::collections::HashMap; use std::io::Read as _; use std::os::fd::AsRawFd as _; use std::os::unix::fs::PermissionsExt as _; + #[test] + fn comm_target_accepts_only_the_callers_own_thread_names() { + let (caller_tid, group) = (4242, 4200); + for (path, expected) in [ + ("/proc/thread-self/comm", Some(caller_tid)), + ("/proc/self/comm", Some(group)), + ("/proc/4200/comm", Some(group)), + ("/proc/self/task/4243/comm", Some(4243)), + ("/proc/4200/task/4243/comm", Some(4243)), + // Another process's thread, or not a comm file. + ("/proc/1/task/1/comm", None), + ("/proc/9999/comm", None), + ("/proc/self/task/4243/environ", None), + ("/proc/self/task/4243/comm/extra", None), + ("/proc/self/mem", None), + ] { + assert_eq!(comm_target(path, caller_tid, group), expected, "{path}"); + } + } + #[test] fn paths_cannot_escape_the_managed_tree() { let valid = "/run/openshell/providers/acme/client.toml"; diff --git a/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs b/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs index 81d0b1630f..e9fa55cc1b 100644 --- a/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs +++ b/crates/openshell-sandbox/src/sandbox/linux/seccomp.rs @@ -5,7 +5,11 @@ //! //! The filter uses a default-allow policy with targeted blocks: //! -//! 1. **Socket domain blocks** -- prevent raw/kernel sockets that bypass the proxy +//! 1. **Socket domain allowlist** -- only `AF_UNIX`, `AF_NETLINK`, and (when +//! networking is enabled) the brokered `AF_INET`/`AF_INET6` families can be +//! created. Every other family is refused, because protocol families such as +//! `AF_RXRPC`, `AF_SMC`, and `AF_KCM` carry traffic over kernel-owned +//! sockets that the broker never creates or confines to loopback. //! 2. **Unconditional syscall blocks** -- block syscalls that enable sandbox escape //! (fileless exec, ptrace, BPF, cross-process memory access, `io_uring`, mount) //! 3. **Conditional syscall blocks** -- block dangerous flag combinations on otherwise @@ -184,23 +188,18 @@ fn apply_runtime_filters( fn build_filter_rules(allow_inet: bool) -> Result>> { let mut rules: BTreeMap> = BTreeMap::new(); - // --- Socket domain blocks --- - let mut blocked_domains = vec![ - libc::AF_PACKET, - libc::AF_BLUETOOTH, - libc::AF_VSOCK, - // AF_NETLINK is handled separately below: NETLINK_ROUTE (protocol 0) - // is allowed for getifaddrs(3); all other netlink protocols are blocked. - ]; - if !allow_inet { - blocked_domains.push(libc::AF_INET); - blocked_domains.push(libc::AF_INET6); - } - - for domain in blocked_domains { - debug!(domain, "Blocking socket domain via seccomp"); - add_socket_domain_rule(&mut rules, domain)?; + // --- Socket domain allowlist --- + // AF_NETLINK is narrowed further below: only NETLINK_ROUTE (protocol 0) + // is allowed, for getifaddrs(3). + let mut allowed_domains = vec![libc::AF_UNIX, libc::AF_NETLINK]; + if allow_inet { + allowed_domains.extend([libc::AF_INET, libc::AF_INET6]); } + debug!(?allowed_domains, "Restricting socket domains via seccomp"); + add_socket_domain_allowlist(&mut rules, libc::SYS_socket, &allowed_domains)?; + // socketpair(2) is only meaningful for AF_UNIX here; other families either + // reject it or create kernel transport sockets. + add_socket_domain_allowlist(&mut rules, libc::SYS_socketpair, &[libc::AF_UNIX])?; // Allow AF_NETLINK only for NETLINK_ROUTE (protocol 0). // @@ -296,14 +295,27 @@ fn build_filter_rules(allow_inet: bool) -> Result Ok(rules) } +/// Refuse `syscall` unless its domain argument is one of `allowed`. +/// +/// A seccomp rule matches only when all of its conditions hold, so one rule +/// with a `!=` condition per allowed domain matches exactly the domains +/// outside the allowlist. The domain is a scalar argument that another thread +/// cannot replace before the kernel reads it. #[allow(clippy::cast_sign_loss)] -fn add_socket_domain_rule(rules: &mut BTreeMap>, domain: i32) -> Result<()> { - let condition = - SeccompCondition::new(0, SeccompCmpArgLen::Dword, SeccompCmpOp::Eq, domain as u64) - .into_diagnostic()?; - - let rule = SeccompRule::new(vec![condition]).into_diagnostic()?; - rules.entry(libc::SYS_socket).or_default().push(rule); +fn add_socket_domain_allowlist( + rules: &mut BTreeMap>, + syscall: i64, + allowed: &[i32], +) -> Result<()> { + let conditions = allowed + .iter() + .map(|domain| { + SeccompCondition::new(0, SeccompCmpArgLen::Dword, SeccompCmpOp::Ne, *domain as u64) + .into_diagnostic() + }) + .collect::>>()?; + let rule = SeccompRule::new(conditions).into_diagnostic()?; + rules.entry(syscall).or_default().push(rule); Ok(()) } @@ -851,6 +863,78 @@ mod tests { ); } + #[test] + fn behavioral_socket_families_are_allowlisted() { + // Applying a filter is irreversible, so run the probe in a fresh copy + // of this test binary rather than in the harness process. + const CHILD_MARKER: &str = "OPENSHELL_SOCKET_FAMILY_ALLOWLIST_CHILD"; + // libc does not export these family numbers. + const AF_KCM: i32 = 41; + const AF_SMC: i32 = 43; + if std::env::var_os(CHILD_MARKER).is_some() { + set_no_new_privs().expect("set no_new_privs"); + apply_filter(&build_filter(true).unwrap()).expect("apply proxy-mode filter"); + let create = |domain: i32, kind: socket2::Type| { + socket2::Socket::new(socket2::Domain::from(domain), kind, None) + .map(drop) + .map_err(|error| error.raw_os_error()) + }; + for (domain, kind) in [ + (libc::AF_UNIX, socket2::Type::STREAM), + (libc::AF_NETLINK, socket2::Type::RAW), + (libc::AF_INET, socket2::Type::STREAM), + (libc::AF_INET6, socket2::Type::DGRAM), + ] { + assert_eq!( + create(domain, kind), + Ok(()), + "domain {domain} must be allowed" + ); + } + // Families whose kernel transport sockets the broker cannot + // confine, plus previously denylisted ones. + for (domain, kind) in [ + (libc::AF_RXRPC, socket2::Type::DGRAM), + (AF_SMC, socket2::Type::STREAM), + (AF_KCM, socket2::Type::DGRAM), + (libc::AF_ALG, socket2::Type::SEQPACKET), + (libc::AF_TIPC, socket2::Type::from(libc::SOCK_RDM)), + (libc::AF_PACKET, socket2::Type::RAW), + (libc::AF_VSOCK, socket2::Type::STREAM), + ] { + assert_eq!( + create(domain, kind), + Err(Some(libc::EPERM)), + "domain {domain} must be refused by the filter" + ); + } + assert!( + socket2::Socket::pair(socket2::Domain::UNIX, socket2::Type::STREAM, None).is_ok() + ); + assert_eq!( + socket2::Socket::pair( + socket2::Domain::from(libc::AF_TIPC), + socket2::Type::STREAM, + None + ) + .map(drop) + .map_err(|error| error.raw_os_error()), + Err(Some(libc::EPERM)) + ); + return; + } + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "sandbox::linux::seccomp::tests::behavioral_socket_families_are_allowlisted", + "--nocapture", + ]) + .env(CHILD_MARKER, "1") + .status() + .expect("run isolated socket family test"); + assert!(status.success(), "isolated socket family test failed"); + } + #[test] fn behavioral_block_mode_denies_inet_and_packet_sockets() { let filter = build_filter(false).unwrap(); diff --git a/docs/about/support-matrix.mdx b/docs/about/support-matrix.mdx index 4115854e8e..c62d401183 100644 --- a/docs/about/support-matrix.mdx +++ b/docs/about/support-matrix.mdx @@ -170,8 +170,8 @@ when it runs inside a container or microVM: | -------------------------------------------------------------- | ----------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | [Landlock LSM](https://docs.kernel.org/security/landlock.html) | Required | ABI 3 or newer, introduced in Linux 6.2, with Landlock enabled. The mandatory baseline protects private channel and bootstrap files, including against truncation. A filesystem policy's `best_effort` setting never disables this baseline. | | seccomp | Required | Nested user-notification filters and atomic `SECCOMP_IOCTL_NOTIF_ADDFD` with `SECCOMP_ADDFD_FLAG_SEND`, usable under the runtime's existing seccomp profile without added capabilities. The sandbox actively probes these operations before admitting the workload. | -| Task-memory access | Required | The non-dumpable broker must be able to read and write a same-UID, dumpable workload child's memory through `process_vm_readv` / `process_vm_writev` or `/proc//mem`. The sandbox actively probes the production parent-to-child topology before admitting the workload. | -| seccomp `WAIT_KILLABLE_RECV` | Recommended (Linux 5.19+) | Keeps a notified workload thread in a kill-only wait so the broker can safely write mediated results into workload memory. Without it (kernels < 5.19, for example RHEL 9.x / RHCOS 5.14) the sandbox still starts, in a reduced **legacy read-only** mode described below. | +| Task-memory access | Required | The non-dumpable broker must be able to read a same-UID, dumpable workload child's memory through `process_vm_readv` or `/proc//mem`. The broker never writes workload memory. The sandbox actively probes the production parent-to-child topology before admitting the workload. | +| Socket device binding | Required | The sandbox binds every workload TCP and UDP socket to the loopback interface with `SO_BINDTODEVICE`, without added capabilities. Accepted sockets inherit the binding, so workloads accept connections only from inside the sandbox network namespace. The sandbox probes this before admitting the workload. | A kernel version alone does not establish support. A disabled Landlock LSM or a runtime profile that blocks the required seccomp operations causes launch to @@ -179,33 +179,9 @@ fail closed. An upstream Linux 6.2 or newer kernel provides the required Landlock ABI; distribution backports must pass the same active qualification. The broker remains non-dumpable during qualification. A runtime may satisfy task-memory access through `/proc//mem` even when its kernel omits the -`process_vm_readv` and `process_vm_writev` system calls; OpenShell qualifies the +`process_vm_readv` system call; OpenShell qualifies the same parent-to-executed-child access shape used by mediated workloads. -### Legacy read-only mode (kernels before Linux 5.19) - -`SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` was added in Linux 5.19. On older -kernels — notably RHEL 9.x and RHCOS, which ship a 5.14 kernel — the sandbox -cannot install a kill-only listener, so it falls back to a plain listener and -runs in a **legacy read-only** cancellation mode. The sandbox starts and -enforces the full isolation boundary (Landlock, the outer NetworkPolicy fence, -DNS and TCP authorization); the only difference is that the broker refuses the -mediated operations that write results back into workload memory, failing them -closed with `EOPNOTSUPP`: - -- `getpeername`; -- `accept` / `accept4` **when a non-null peer-address argument is supplied** - (a null address argument still works); -- `sendmmsg` paths that write per-message lengths back to the caller. - -Socket creation, `connect`, `bind`, `listen`, `sendto`, and `sendmsg` are -unaffected. Outbound-oriented workloads generally run unchanged; server -workloads whose accept wrappers request the peer address will see `EOPNOTSUPP` -until the node runs a kernel that provides `WAIT_KILLABLE_RECV` (Linux 5.19+, or -a distribution backport). The selected mode is reported in the sandbox -qualification output as `seccomp_listener_mode` (`killable` or -`legacy_read_only`). - On macOS, these kernel modules run inside the Docker Desktop Linux VM, not on the host kernel. ## Agent Workloads diff --git a/docs/kubernetes/openshift.mdx b/docs/kubernetes/openshift.mdx index c6189b1ebb..7e7887f092 100644 --- a/docs/kubernetes/openshift.mdx +++ b/docs/kubernetes/openshift.mdx @@ -19,20 +19,11 @@ process to install a nested seccomp user-notification filter and use Landlock. OpenShell fails sandbox startup when either capability-free runtime probe fails. -## Node kernel and legacy read-only mode - -OpenShift nodes run RHCOS, which currently ships a RHEL 9.x kernel (5.14). That -kernel predates `SECCOMP_FILTER_FLAG_WAIT_KILLABLE_RECV` (Linux 5.19), so the -sandbox starts in a reduced **legacy read-only** cancellation mode. Isolation is -unchanged, but the broker fails closed with `EOPNOTSUPP` on the mediated -operations that write results back into workload memory — `getpeername`, -`accept`/`accept4` with a non-null peer-address argument, and `sendmmsg` -per-message length write-backs. Outbound-oriented workloads run unchanged; -server workloads that read the peer address on accept need a node kernel with -`WAIT_KILLABLE_RECV` (Linux 5.19+, or a distribution backport). See the -[support matrix](/about/support-matrix#legacy-read-only-mode-kernels-before-linux-519) -for the full behavior; the selected mode is reported as `seccomp_listener_mode` -in the sandbox qualification output. +## Node kernel + +OpenShell requires OpenShift 4.19 or later. The RHCOS kernels in OpenShift 4.16 +through 4.18 (RHEL 9.4, 5.14.0-427) are built without Landlock, so sandbox +startup fails its Landlock probe on those releases. ## Prerequisites diff --git a/docs/security/best-practices.mdx b/docs/security/best-practices.mdx index 2f5b346708..e70f38ecc4 100644 --- a/docs/security/best-practices.mdx +++ b/docs/security/best-practices.mdx @@ -220,7 +220,7 @@ OpenShell applies seccomp in two phases. A narrow supervisor-startup prelude run | Aspect | Detail | |---|---| | Startup prelude | After privileged bootstrap helpers complete, including network setup and provider-token SPIFFE child mount-namespace preparation, the supervisor sets `PR_SET_NO_NEW_PRIVS` and synchronizes a seccomp filter across all runtime threads that blocks `mount`, the new mount API syscalls, `pivot_root`, `umount2`, `bpf`, `perf_event_open`, `userfaultfd`, module-loading syscalls, and kexec. This closes the long-lived privileged remount and kernel-surface window while leaving required setup syscalls such as `setns` available. | -| Socket domains | The filter allows `AF_INET` and `AF_INET6` (for proxy communication) and blocks `AF_PACKET`, `AF_BLUETOOTH`, and `AF_VSOCK` with `EPERM`. `AF_NETLINK` is partially allowed: only `NETLINK_ROUTE` (protocol 0) is permitted so that `getifaddrs(3)` works; all other netlink protocols are blocked. Write operations via `NETLINK_ROUTE` still require `CAP_NET_ADMIN`, which the sandbox does not grant. | +| Socket domains | The filter allows only `AF_UNIX`, `AF_NETLINK`, and the brokered `AF_INET` and `AF_INET6` families. Every other family, including `AF_PACKET`, `AF_VSOCK`, `AF_RXRPC`, `AF_SMC`, `AF_KCM`, and `AF_ALG`, fails with `EPERM`, and `socketpair(2)` is limited to `AF_UNIX`. `AF_NETLINK` is partially allowed: only `NETLINK_ROUTE` (protocol 0) is permitted so that `getifaddrs(3)` works; all other netlink protocols are blocked. Write operations via `NETLINK_ROUTE` still require `CAP_NET_ADMIN`, which the sandbox does not grant. | | Runtime unconditional syscall blocks | `memfd_create`, `ptrace`, `bpf`, `process_vm_readv`, `process_vm_writev`, `pidfd_open`, `pidfd_getfd`, `pidfd_send_signal`, `io_uring_setup`, `mount`, `fsopen`, `fsconfig`, `fsmount`, `fspick`, `move_mount`, `open_tree`, `setns`, `umount2`, `pivot_root`, `userfaultfd`, `perf_event_open`. | | Conditional syscall blocks | `execveat` with `AT_EMPTY_PATH`, `unshare` and `clone` with `CLONE_NEWUSER`, and `seccomp(SECCOMP_SET_MODE_FILTER)` are denied with `EPERM`. | | What you can change | This is not a user-facing knob. OpenShell enforces it automatically. | diff --git a/e2e/rust/src/harness/sandbox.rs b/e2e/rust/src/harness/sandbox.rs index 9922db02b4..0ebfde9081 100644 --- a/e2e/rust/src/harness/sandbox.rs +++ b/e2e/rust/src/harness/sandbox.rs @@ -618,7 +618,11 @@ impl SandboxGuard { for arg in argv { cmd.arg(arg); } - cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); + // Never share the test runner's stdin: parallel test processes share + // its open file description, and the command needs no input. + cmd.stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); let output = cmd .output()