diff --git a/litebox/src/broker/mod.rs b/litebox/src/broker/mod.rs index 20d58a54c..100907163 100644 --- a/litebox/src/broker/mod.rs +++ b/litebox/src/broker/mod.rs @@ -19,13 +19,13 @@ use litebox_broker_protocol::fs::{ }; use litebox_broker_protocol::pipe::{CreatePipeResponse, MAX_PIPE_TRANSFER_SIZE}; use litebox_broker_protocol::process::{ - CreatedProcess, MAX_CHILD_MEMORY_WRITE_SIZE, MAX_CHILD_OBJECT_DUPLICATES, - MAX_PROCESS_BOOTSTRAP_SIZE, ProcessExitStatus, ProcessTermination, + ChildExit, ChildSelector, MAX_CHILD_MEMORY_WRITE_SIZE, MAX_CHILD_OBJECT_DUPLICATES, + MAX_PROCESS_BOOTSTRAP_SIZE, ProcessExitStatus, ProcessIdentity, ProcessInfo, }; use litebox_broker_protocol::random::MAX_RANDOM_TRANSFER_SIZE; use litebox_broker_protocol::readiness::ReadinessFlags; use litebox_broker_protocol::shared_buffer::SHARED_BUFFER_SLOT_SIZE; -use litebox_broker_protocol::signal::PendingSignal; +use litebox_broker_protocol::signal::{SignalEvent, SignalTarget}; use litebox_broker_protocol::socket::{ AcceptSocketResponse, MAX_SOCKET_TRANSFER_SIZE, MAX_UDP_DATAGRAM_SIZE, ReceiveFlags as BrokerReceiveFlags, ReceiveFromFlags as BrokerReceiveFromFlags, @@ -54,7 +54,12 @@ use shared_buffer::{AcquireError, SlotAllocator, SlotLease}; /// Longer-term broker integrations should move away from blocking control calls /// once the local-core wait and notification model supports that shape. pub(crate) trait BrokerControl: Send + Sync { - fn allocate_child_process(&self) -> core::result::Result; + fn allocate_child_process(&self) -> core::result::Result; + + fn cancel_child_process( + &self, + child_process_id: litebox_broker_protocol::ProcessId, + ) -> core::result::Result<(), BrokerControlError>; fn start_child_process( &self, @@ -88,23 +93,41 @@ pub(crate) trait BrokerControl: Send + Sync { fn set_child_reaping(&self, enabled: bool) -> core::result::Result<(), BrokerControlError>; - fn process_exit_status( + fn set_orphan_adoption(&self, enabled: bool) -> core::result::Result<(), BrokerControlError>; + + fn reap_child( &self, - handle: ObjectHandle, - ) -> core::result::Result; + selector: ChildSelector, + ) -> core::result::Result; + + fn process_info( + &self, + process_id: litebox_broker_protocol::ProcessId, + ) -> core::result::Result; fn open_signals(&self) -> core::result::Result; fn send_signal( &self, - process_id: litebox_broker_protocol::ProcessId, + target: SignalTarget, signal: u32, ) -> core::result::Result<(), BrokerControlError>; fn take_signal( &self, handle: ObjectHandle, - ) -> core::result::Result; + ) -> core::result::Result; + + fn set_process_group( + &self, + process_id: litebox_broker_protocol::ProcessId, + process_group: litebox_broker_protocol::ProcessId, + ) -> core::result::Result<(), BrokerControlError>; + + fn create_session( + &self, + process_id: litebox_broker_protocol::ProcessId, + ) -> core::result::Result<(), BrokerControlError>; fn create_thread(&self) -> core::result::Result; @@ -512,10 +535,17 @@ where Platform: RawSyncPrimitivesProvider + TimeProvider, Channel: LocalCallChannel + Send + Sync, { - fn allocate_child_process(&self) -> core::result::Result { + fn allocate_child_process(&self) -> core::result::Result { self.request(BrokerLocal::allocate_child_process) } + fn cancel_child_process( + &self, + child_process_id: litebox_broker_protocol::ProcessId, + ) -> core::result::Result<(), BrokerControlError> { + self.request(|local| local.cancel_child_process(child_process_id)) + } + fn start_child_process( &self, child_process_id: litebox_broker_protocol::ProcessId, @@ -581,11 +611,22 @@ where self.request(|local| local.set_child_reaping(enabled)) } - fn process_exit_status( + fn set_orphan_adoption(&self, enabled: bool) -> core::result::Result<(), BrokerControlError> { + self.request(|local| local.set_orphan_adoption(enabled)) + } + + fn reap_child( &self, - handle: ObjectHandle, - ) -> core::result::Result { - self.request(|local| local.process_exit_status(handle)) + selector: ChildSelector, + ) -> core::result::Result { + self.request(|local| local.reap_child(selector)) + } + + fn process_info( + &self, + process_id: litebox_broker_protocol::ProcessId, + ) -> core::result::Result { + self.request(|local| local.process_info(process_id)) } fn open_signals(&self) -> core::result::Result { @@ -594,19 +635,34 @@ where fn send_signal( &self, - process_id: litebox_broker_protocol::ProcessId, + target: SignalTarget, signal: u32, ) -> core::result::Result<(), BrokerControlError> { - self.request(|local| local.send_signal(process_id, signal)) + self.request(|local| local.send_signal(target, signal)) } fn take_signal( &self, handle: ObjectHandle, - ) -> core::result::Result { + ) -> core::result::Result { self.request(|local| local.take_signal(handle)) } + fn set_process_group( + &self, + process_id: litebox_broker_protocol::ProcessId, + process_group: litebox_broker_protocol::ProcessId, + ) -> core::result::Result<(), BrokerControlError> { + self.request(|local| local.set_process_group(process_id, process_group)) + } + + fn create_session( + &self, + process_id: litebox_broker_protocol::ProcessId, + ) -> core::result::Result<(), BrokerControlError> { + self.request(|local| local.create_session(process_id)) + } + fn create_thread(&self) -> core::result::Result { self.request(BrokerLocal::create_thread) } diff --git a/litebox/src/fs/file.rs b/litebox/src/fs/file.rs index fab6de0aa..788d1b8e8 100644 --- a/litebox/src/fs/file.rs +++ b/litebox/src/fs/file.rs @@ -76,7 +76,7 @@ impl LiteBox } /// Returns a descriptor for a file this process inherited from its parent - /// through [`Process::inherit`](crate::process::Process::inherit). + /// through [`PendingChild::inherit`](crate::process::PendingChild::inherit). /// /// The descriptor owns `handle`, so callers adopt each handle once and /// duplicate the descriptor for every other use. diff --git a/litebox/src/pipes.rs b/litebox/src/pipes.rs index 35e93e3e7..3140827b2 100644 --- a/litebox/src/pipes.rs +++ b/litebox/src/pipes.rs @@ -401,7 +401,7 @@ impl LiteBox { } /// Returns a descriptor for a pipe end this process inherited from its - /// parent through [`Process::inherit`](crate::process::Process::inherit). + /// parent through [`PendingChild::inherit`](crate::process::PendingChild::inherit). /// /// The descriptor owns `handle`, so callers adopt each handle once and /// duplicate the descriptor for every other use. The end has no peer in diff --git a/litebox/src/process.rs b/litebox/src/process.rs index b63a989d3..d6cf3f12a 100644 --- a/litebox/src/process.rs +++ b/litebox/src/process.rs @@ -1,17 +1,19 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT license. -//! Broker-backed guest process creation, child termination, and signals -//! between processes. +//! Broker-backed guest process creation, the process tree, process groups +//! and sessions, and signals between processes. use alloc::sync::Arc; use alloc::vec::Vec; +use core::sync::atomic::{AtomicBool, Ordering}; use litebox_broker_protocol::error::ErrorCode; use litebox_broker_protocol::process::{ - MAX_CHILD_MEMORY_WRITE_SIZE, ProcessExitStatus, ProcessIdentity, ProcessTermination, + ChildExit, ChildSelector, MAX_CHILD_MEMORY_WRITE_SIZE, ProcessExitStatus, ProcessIdentity, + ProcessInfo, }; -use litebox_broker_protocol::signal::PendingSignal; +use litebox_broker_protocol::signal::{SignalEvent, SignalTarget}; use litebox_broker_protocol::{ObjectHandle, ProcessId}; use litebox_platform::time::TimeProvider; @@ -33,8 +35,8 @@ pub enum ProcessError { /// The broker association failed. #[error("process service failed")] ServiceFailed, - /// Process duplication is disabled by policy. - #[error("process duplication is denied")] + /// The operation is denied by policy or by process group rules. + #[error("process operation is denied")] PolicyDenied, /// Another pending child process already exists. #[error("a child process is already pending")] @@ -58,14 +60,18 @@ pub enum ProcessError { impl LiteBox { /// Allocates one pending child process. - pub fn allocate_child_process(&self) -> Result, ProcessError> { + pub fn allocate_child_process(&self) -> Result { let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; - let child = broker.allocate_child_process()?; - Ok(Process::new(self, broker, child.identity, child.handle)) + let identity = broker.allocate_child_process()?; + Ok(PendingChild { + broker, + identity, + pending: AtomicBool::new(true), + }) } /// Reports this process's final termination status, which its parent - /// observes once this process's runner exits. + /// reaps once this process's runner exits. pub fn report_exit_status(&self, exit_status: ProcessExitStatus) -> Result<(), ProcessError> { let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; Ok(broker.report_exit_status(exit_status)?) @@ -80,23 +86,80 @@ impl LiteBox { Ok(broker.set_child_reaping(enabled)?) } - /// Sends `signal` to process `process_id`, or only checks that the process - /// exists if `signal` is zero. + /// Sets whether this process adopts the children of its exiting + /// descendants. /// - /// The target takes the signal through [`Signals`]. A signal already - /// pending for the target is not sent again. - pub fn send_signal(&self, process_id: ProcessId, signal: u32) -> Result<(), ProcessError> { + /// The children of an exiting process move to its nearest running ancestor + /// that adopts them, or are left without a parent, which reaps each as it + /// terminates. + pub fn set_orphan_adoption(&self, enabled: bool) -> Result<(), ProcessError> { let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; - match broker.send_signal(process_id, signal) { - Err(BrokerControlError::Broker(ErrorCode::UnknownObject)) => { - Err(ProcessError::NoSuchProcess) - } - result => Ok(result?), + Ok(broker.set_orphan_adoption(enabled)?) + } + + /// Reaps the oldest of this process's terminated children that `selector` + /// matches, in the order this process gained them, returning its exit. + /// + /// Returns `None` if every matching child is live, and + /// [`ProcessError::NoSuchProcess`] if no child matches. Each child that + /// terminates, or leaves without terminating, such as a child whose + /// startup failed, is also reported through [`Signals`]. + pub fn reap_child(&self, selector: ChildSelector) -> Result, ProcessError> { + let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; + match broker.reap_child(selector) { + Ok(exit) => Ok(Some(exit)), + Err(BrokerControlError::Broker(ErrorCode::WouldBlock)) => Ok(None), + Err(error) => Err(no_such_process(error)), } } - /// Opens the signals other processes send to this process, including - /// those sent before it is opened. + /// Returns the place of process `process_id` in the process tree. + pub fn process_info(&self, process_id: ProcessId) -> Result { + let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; + broker.process_info(process_id).map_err(no_such_process) + } + + /// Sends `signal` to the processes `target` selects, or only checks that + /// one exists if `signal` is zero. + /// + /// Each target takes the signal through [`Signals`]. A signal already + /// pending for a target is not sent again. Returns + /// [`ProcessError::NoSuchProcess`] if no process is targeted. + pub fn send_signal(&self, target: SignalTarget, signal: u32) -> Result<(), ProcessError> { + let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; + broker.send_signal(target, signal).map_err(no_such_process) + } + + /// Moves process `process_id`, which is this process or one of its + /// children, into `process_group`, creating the group if it is + /// `process_id`. + /// + /// Returns [`ProcessError::PolicyDenied`] if the process is in another + /// session than this process or leads a session, or if `process_group` is + /// neither `process_id` nor an existing group in this process's session. + pub fn set_process_group( + &self, + process_id: ProcessId, + process_group: ProcessId, + ) -> Result<(), ProcessError> { + let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; + broker + .set_process_group(process_id, process_group) + .map_err(no_such_process) + } + + /// Makes process `process_id`, which is this process or its pending + /// child, the leader of a new session and of a new process group in it. + /// + /// Returns [`ProcessError::PolicyDenied`] if a process group already has + /// the ID `process_id`. + pub fn create_session(&self, process_id: ProcessId) -> Result<(), ProcessError> { + let broker = self.broker_control().ok_or(ProcessError::Unavailable)?; + broker.create_session(process_id).map_err(no_such_process) + } + + /// Opens the signals other processes send to this process and the exits + /// of its children, including those before it is opened. /// /// A process may hold only one [`Signals`] at a time. pub fn open_signals(&self) -> Result, ProcessError> { @@ -106,7 +169,7 @@ impl LiteBox { } } -/// A descriptor a pending child process can inherit through [`Process::inherit`]. +/// A descriptor a pending child process can inherit through [`PendingChild::inherit`]. /// /// This identifies only the object. The shim records guest-specific details, such as the /// descriptor number, in the child's startup payload alongside the handle `inherit` returns. @@ -117,50 +180,18 @@ pub enum InheritableFd { Pipe(Arc>), } -/// Termination state of a child process. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ChildStatus { - /// The process has not terminated. - Live, - /// The process terminated with this status and waits to be reported. - Terminated(ProcessExitStatus), - /// The process terminated with this status and was reaped as it - /// terminated, so no wait reports it. - Reaped(ProcessExitStatus), -} - -/// A broker process object. +/// A child process that has not started. /// -/// It reports [`Events::IN`] once the process terminates. Dropping it closes -/// its handle, releases the process's retained exit status, and wakes its -/// observers to recheck their state. -pub struct Process { +/// Dropping it before it starts or exits discards the child, as if it had +/// never been created, except that its parent learns it was removed. +pub struct PendingChild { broker: Arc, identity: ProcessIdentity, - handle: ObjectHandle, - pollable_registry: Arc>, - pollee: Arc>, + /// Cleared once the child starts or exits. + pending: AtomicBool, } -impl Process { - fn new( - litebox: &LiteBox, - broker: Arc, - identity: ProcessIdentity, - handle: ObjectHandle, - ) -> Self { - let pollable_registry = litebox.broker_pollable_registry(); - let pollee = Arc::new(Pollee::new()); - pollable_registry.register_pollable(handle, &pollee); - Self { - broker, - identity, - handle, - pollable_registry, - pollee, - } - } - +impl PendingChild { /// Returns the broker-assigned process and initial thread IDs. pub fn identity(&self) -> ProcessIdentity { self.identity @@ -168,9 +199,10 @@ impl Process { /// Starts this pending child process in a fresh runner. pub fn start(&self, payload: &[u8]) -> Result<(), ProcessError> { - Ok(self - .broker - .start_child_process(self.identity.process_id, payload)?) + self.broker + .start_child_process(self.identity.process_id, payload)?; + self.pending.store(false, Ordering::Relaxed); + Ok(()) } /// Gives this pending child process its own references to the objects @@ -186,7 +218,7 @@ impl Process { /// # Panics /// /// Panics if the broker returns fewer handles than were duplicated. - pub fn inherit( + pub fn inherit( &self, litebox: &LiteBox, fds: &[InheritableFd], @@ -252,53 +284,26 @@ impl Process { /// Records that this pending child process exited without starting a /// runner, leaving it a zombie reporting `exit_status`. pub fn exit(&self, exit_status: ProcessExitStatus) -> Result<(), ProcessError> { - Ok(self - .broker - .exit_child_process(self.identity.process_id, exit_status)?) - } - - /// Returns the process's termination state. - pub fn status(&self) -> Result { - match self.broker.process_exit_status(self.handle) { - Ok(ProcessTermination { - exit_status, - reaped: false, - }) => Ok(ChildStatus::Terminated(exit_status)), - Ok(ProcessTermination { - exit_status, - reaped: true, - }) => Ok(ChildStatus::Reaped(exit_status)), - Err(BrokerControlError::Broker(ErrorCode::WouldBlock)) => Ok(ChildStatus::Live), - Err(error) => Err(error.into()), - } + self.broker + .exit_child_process(self.identity.process_id, exit_status)?; + self.pending.store(false, Ordering::Relaxed); + Ok(()) } } -impl Drop for Process { +impl Drop for PendingChild { fn drop(&mut self) { - self.pollable_registry.unregister_pollable(self.handle); - let _ = self.broker.close_object(self.handle); - // Another waiter may still be blocked on this process's exit - // notification, which is discarded once the handle is unregistered. - self.pollee.wake_observers(); - } -} - -impl IOPollable for Process { - fn register_observer(&self, observer: alloc::sync::Weak>, mask: Events) { - self.pollee.register_observer(observer, mask); - } - - fn check_io_events(&self) -> Events { - self.broker - .check_readiness(self.handle) - .map_or(Events::ERR, readiness_events) + if self.pending.load(Ordering::Relaxed) { + // Failure means the child is already gone or the process service failed. + let _ = self.broker.cancel_child_process(self.identity.process_id); + } } } -/// The signals other processes send to this process. +/// The signals other processes send to this process, and the exits of its +/// children. /// -/// It reports [`Events::IN`] while a signal is pending. +/// It reports [`Events::IN`] while either is pending. pub struct Signals { broker: Arc, handle: ObjectHandle, @@ -323,11 +328,16 @@ impl Signals { } } - /// Takes the lowest-numbered pending signal, or returns `None` if none is + /// Takes the lowest-numbered pending signal, or else a pending child + /// exit, or else a pending child removal, or returns `None` if none is /// pending. - pub fn take(&self) -> Result, ProcessError> { + /// + /// Child events are coalesced: while one is pending, later events of its + /// kind are not reported, so a taker reaps children until none has + /// terminated. + pub fn take(&self) -> Result, ProcessError> { match self.broker.take_signal(self.handle) { - Ok(signal) => Ok(Some(signal)), + Ok(event) => Ok(Some(event)), Err(BrokerControlError::Broker(ErrorCode::WouldBlock)) => Ok(None), Err(error) => Err(error.into()), } @@ -353,6 +363,14 @@ impl IOPollable for Signals< } } +/// Converts an error from a request that targets processes by ID. +fn no_such_process(error: BrokerControlError) -> ProcessError { + match error { + BrokerControlError::Broker(ErrorCode::UnknownObject) => ProcessError::NoSuchProcess, + error => error.into(), + } +} + impl From for ProcessError { fn from(error: BrokerControlError) -> Self { match error { @@ -374,12 +392,13 @@ impl From for ProcessError { #[cfg(test)] mod tests { - use core::sync::atomic::{AtomicBool, Ordering}; + use core::sync::atomic::AtomicUsize; use litebox_broker_local::test_support::test_broker_local; use litebox_broker_protocol::message::{ BrokerOperation, BrokerRequest, BrokerResponse, BrokerResult, }; + use litebox_broker_protocol::process::{CreateThreadRequest, CreateThreadResponse}; use litebox_broker_protocol::shared_buffer::SHARED_BUFFER_POOL_SIZE; use litebox_broker_protocol::{ProcessId, ThreadId}; use litebox_broker_transport::channel::LocalCallChannel; @@ -389,51 +408,49 @@ mod tests { use crate::platform::mock::MockPlatform; #[test] - fn dropping_process_wakes_observers() { + fn dropping_a_pending_child_cancels_it() { + let cancels = Arc::new(AtomicUsize::new(0)); let litebox = LiteBox::new_with_broker_local( MockPlatform::new(), - test_broker_local(CloseChannel, Arc::new(NoopSharedMemory)), - ); - let process = Process::new( - &litebox, - litebox.broker_control().unwrap(), - ProcessIdentity { - process_id: ProcessId(3), - initial_thread_id: ThreadId(4), - }, - ObjectHandle(7), + test_broker_local(ChildChannel(cancels.clone()), Arc::new(NoopSharedMemory)), ); - let observer = Arc::new(WakeObserver(AtomicBool::new(false))); - process.register_observer(Arc::downgrade(&observer) as _, Events::IN); - - // A waiter that reaps this process may drop it before the exit - // notification arrives, so other waiters must still wake. - drop(process); - - assert!(observer.0.load(Ordering::SeqCst)); - } - struct WakeObserver(AtomicBool); + drop(litebox.allocate_child_process().unwrap()); + assert_eq!(cancels.load(Ordering::SeqCst), 1); - impl Observer for WakeObserver { - fn on_events(&self, _events: &Events) { - self.0.store(true, Ordering::SeqCst); - } + let child = litebox.allocate_child_process().unwrap(); + child.exit(ProcessExitStatus::Exited { code: 0 }).unwrap(); + drop(child); + assert_eq!(cancels.load(Ordering::SeqCst), 1); } - struct CloseChannel; + /// Serves one pending child at a time, counting its cancellations. + struct ChildChannel(Arc); - impl LocalCallChannel for CloseChannel { + impl LocalCallChannel for ChildChannel { type Error = (); fn call(&self, request: BrokerRequest) -> Result { - assert_eq!( - request.operation, - BrokerOperation::CloseObject(ObjectHandle(7)) - ); + let child = ProcessId(3); + let result = match request.operation { + BrokerOperation::CreateThread(CreateThreadRequest::Process) => { + BrokerResult::CreateThread(CreateThreadResponse::Process(ProcessIdentity { + process_id: child, + initial_thread_id: ThreadId(4), + })) + } + BrokerOperation::ExitChildProcess(request) if request.child_process_id == child => { + BrokerResult::ProcessExited + } + BrokerOperation::CancelChildProcess(process_id) if process_id == child => { + self.0.fetch_add(1, Ordering::SeqCst); + BrokerResult::ChildProcessCancelled + } + operation => panic!("unexpected operation {operation:?}"), + }; Ok(BrokerResponse { request_id: request.request_id, - result: BrokerResult::ObjectClosed, + result, }) } } diff --git a/litebox_broker_core/src/id.rs b/litebox_broker_core/src/id.rs index e85ea2b57..f7f21b384 100644 --- a/litebox_broker_core/src/id.rs +++ b/litebox_broker_core/src/id.rs @@ -3,7 +3,7 @@ //! Shared broker process and thread ID allocation. -use hashbrown::HashSet; +use hashbrown::HashMap; use crate::{BrokerError, Result}; @@ -17,7 +17,9 @@ pub(crate) const MAX_ALLOCATED_ID: u32 = 0x3fff_fffe; pub(crate) struct IdAllocator { next: u32, max_id: u32, - occupied: HashSet, + /// The number of users of each allocated ID, which is reused only once + /// it has none. + occupied: HashMap, failed: bool, } @@ -29,11 +31,12 @@ impl IdAllocator { Ok(Self { next: 1, max_id, - occupied: HashSet::new(), + occupied: HashMap::new(), failed: false, }) } + /// Allocates an unused ID with one user. pub(crate) fn allocate(&mut self) -> Result { if self.failed || self.occupied.len() >= self.max_id as usize { return Err(BrokerError::ResourceExhausted); @@ -50,7 +53,8 @@ impl IdAllocator { } else { candidate + 1 }; - if self.occupied.insert(candidate) { + if !self.occupied.contains_key(&candidate) { + self.occupied.insert(candidate, 1); return Ok(candidate); } if self.next == first { @@ -59,9 +63,23 @@ impl IdAllocator { } } + /// Adds a user of the allocated ID `id`. + pub(crate) fn retain(&mut self, id: u32) { + match self.occupied.get_mut(&id) { + Some(users) => *users += 1, + None => self.failed = true, + } + } + + /// Removes a user of `id`, which becomes free for reuse once it has none. pub(crate) fn release(&mut self, id: u32) { - if !self.occupied.remove(&id) { + let Some(users) = self.occupied.get_mut(&id) else { self.failed = true; + return; + }; + *users -= 1; + if *users == 0 { + self.occupied.remove(&id); } } } @@ -84,6 +102,21 @@ mod tests { assert_ne!(first, second); } + #[test] + fn allocator_reuses_an_id_only_after_its_last_user_releases_it() { + let mut allocator = IdAllocator::new(2).unwrap(); + let shared = allocator.allocate().unwrap(); + allocator.retain(shared); + allocator.release(shared); + let other = allocator.allocate().unwrap(); + assert_eq!(allocator.allocate(), Err(BrokerError::ResourceExhausted)); + + allocator.release(shared); + + assert_eq!(allocator.allocate().unwrap(), shared); + assert_ne!(shared, other); + } + #[test] fn allocator_rejects_ids_outside_the_guest_range() { assert!(matches!( diff --git a/litebox_broker_core/src/lib.rs b/litebox_broker_core/src/lib.rs index e564fb7c2..26cc71411 100644 --- a/litebox_broker_core/src/lib.rs +++ b/litebox_broker_core/src/lib.rs @@ -26,6 +26,7 @@ mod object; pub mod pipe; mod policy; mod process; +pub mod process_group; pub mod random; pub mod readiness; pub mod signal; @@ -41,6 +42,7 @@ pub mod test_support; use alloc::sync::{Arc, Weak}; use core::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}; +use alloc::vec::Vec; use hashbrown::HashMap; use litebox_broker_protocol::{ObjectHandle, ProcessId}; use spin::{Mutex, rwlock::RwLock}; @@ -234,6 +236,13 @@ pub struct BrokerCore { pub(crate) limits: BrokerCoreLimits, pub(crate) ids: Arc>, pub(crate) processes: Arc>>>, + /// Serializes changes to the process tree, namely parent and child links, + /// exits and reaping, and process group and session membership, and + /// selections over it. + /// + /// Lock order: this lock, then process state locks, then the registry + /// and ID allocator, then process tree links. + pub(crate) process_tree: Arc>, /// Number of broker threads created and not normally retired. pub(crate) active_thread_count: Arc, pub(crate) next_reference_handle: Arc>, @@ -297,6 +306,7 @@ impl BrokerCore { limits, ids: Arc::new(Mutex::new(ids)), processes: Arc::new(RwLock::new(HashMap::new())), + process_tree: Arc::new(Mutex::new(())), active_thread_count: Arc::new(AtomicUsize::new(0)), next_reference_handle: Arc::new(RwLock::new(1)), references: Arc::new(RwLock::new(HashMap::new())), @@ -322,6 +332,32 @@ impl BrokerCore { } } + /// Returns the process `id`, or `UnknownObject` if none exists. + /// + /// A process that was reaped or never started no longer exists. + pub(crate) fn registered_process(&self, id: ProcessId) -> Result> { + // The registry lock is released before the process can drop, since a + // final process drop removes itself from the registry. + let process = self.processes.read().get(&id).and_then(Weak::upgrade); + process + .filter(|process| !process.is_released()) + .ok_or(BrokerError::UnknownObject) + } + + /// Returns every existing process. + pub(crate) fn registered_processes(&self) -> Vec> { + // The registry lock is released before any process can drop, since a + // final process drop removes itself from the registry. + let mut processes = self + .processes + .read() + .values() + .filter_map(Weak::upgrade) + .collect::>(); + processes.retain(|process| !process.is_released()); + processes + } + /// Returns whether any broker process remains registered. #[must_use] pub fn has_processes(&self) -> bool { @@ -360,7 +396,10 @@ impl BrokerCore { Ok((first, second)) } - /// Allocates one authenticated process without an initial thread. + /// Allocates one authenticated process without an initial thread, as a + /// child of the process `parent_id` if given. + /// + /// Returns `PeerClosed` if the parent's owner died or the parent exited. /// /// # Panics /// @@ -371,7 +410,7 @@ impl BrokerCore { caller_credential: CallerCredential, parent_id: Option, ) -> Result> { - let allocate_process = |parent: Option>| { + let allocate_process = |parent: Option<&Arc>| { let mut processes = self.processes.write(); if processes.len() >= self.limits.max_processes { return Err(BrokerError::ResourceExhausted); @@ -379,6 +418,9 @@ impl BrokerCore { processes .try_reserve(1) .map_err(|_| BrokerError::OutOfMemory)?; + if let Some(parent) = parent { + parent.reserve_child()?; + } let raw_id = self.ids.lock().allocate()?; let id = ProcessId(raw_id); let process = Arc::new(BrokerProcess::new( @@ -391,24 +433,24 @@ impl BrokerCore { processes.insert(id, Arc::downgrade(&process)).is_none(), "the ID allocator returned an occupied process ID" ); + if let Some(parent) = parent { + parent.add_child(Arc::clone(&process)); + } Ok(process) }; - if let Some(parent_id) = parent_id { - let parent = self - .processes - .read() - .get(&parent_id) - .and_then(Weak::upgrade) - .ok_or(BrokerError::UnknownObject)?; - return parent.with_live_owner(|| allocate_process(Some(Arc::downgrade(&parent))))?; - } - allocate_process(None) + // The tree lock orders the new link against the parent's exit. + let _tree = self.process_tree.lock(); + let Some(parent_id) = parent_id else { + return allocate_process(None); + }; + let parent = self.registered_process(parent_id)?; + parent.with_live_owner(|| allocate_process(Some(&parent)))? } /// Creates one process and its initial thread. /// - /// If initial-thread creation fails, the process is retired before the + /// If initial-thread creation fails, the process is discarded before the /// error is returned. /// /// # Panics @@ -427,6 +469,7 @@ impl BrokerCore { Ok(process) } Err(error) => { + let _ = process.fail_start(error, false, true); process.retire(true); Err(error) } diff --git a/litebox_broker_core/src/object.rs b/litebox_broker_core/src/object.rs index 6b891bdd4..34e977f10 100644 --- a/litebox_broker_core/src/object.rs +++ b/litebox_broker_core/src/object.rs @@ -13,7 +13,6 @@ use spin::rwlock::RwLock; use crate::event::EventObject; use crate::fs::File; use crate::pipe::PipeObject; -use crate::process::ProcessObject; use crate::readiness::{ReadinessRegistration, ReadinessSink}; use crate::signal::SignalsObject; use crate::socket::SocketObject; @@ -50,7 +49,6 @@ pub(crate) enum ObjectEntry { File(File), Pipe(PipeObject), Socket(SocketObject), - Process(ProcessObject), Signals(SignalsObject), Timer(TimerObject), } @@ -67,7 +65,7 @@ impl ObjectEntry { pub(crate) fn is_duplicable(&self) -> bool { match self { Self::Event(_) | Self::File(_) | Self::Pipe(_) => true, - Self::Socket(_) | Self::Process(_) | Self::Signals(_) | Self::Timer(_) => false, + Self::Socket(_) | Self::Signals(_) | Self::Timer(_) => false, } } @@ -86,11 +84,7 @@ impl ObjectEntry { pipe.watch(®istration)?; Ok(Some(registration)) } - Self::Event(_) - | Self::Socket(_) - | Self::Process(_) - | Self::Signals(_) - | Self::Timer(_) => Ok(None), + Self::Event(_) | Self::Socket(_) | Self::Signals(_) | Self::Timer(_) => Ok(None), } } } @@ -105,7 +99,6 @@ pub(crate) fn readiness(object: &RwLock) -> Result ObjectEntry::Event(event) => return Ok(event.readiness()), ObjectEntry::File(file) => return file.readiness(), ObjectEntry::Pipe(pipe) => return Ok(pipe.readiness()), - ObjectEntry::Process(process) => return Ok(process.readiness()), ObjectEntry::Signals(signals) => return Ok(signals.readiness()), ObjectEntry::Timer(timer) => return Ok(timer.readiness()), ObjectEntry::Socket(socket) => socket.resource(), @@ -124,7 +117,6 @@ pub(crate) fn get_status_flags( ObjectEntry::Pipe(pipe) => return Ok(pipe.get_status_flags()), ObjectEntry::Event(_) | ObjectEntry::Socket(_) - | ObjectEntry::Process(_) | ObjectEntry::Signals(_) | ObjectEntry::Timer(_) => return Err(BrokerError::InvalidRights), }; @@ -146,7 +138,6 @@ pub(crate) fn set_status_flags( } ObjectEntry::Event(_) | ObjectEntry::Socket(_) - | ObjectEntry::Process(_) | ObjectEntry::Signals(_) | ObjectEntry::Timer(_) => return Err(BrokerError::InvalidRights), }; diff --git a/litebox_broker_core/src/process.rs b/litebox_broker_core/src/process.rs index fa60ae05d..01c287218 100644 --- a/litebox_broker_core/src/process.rs +++ b/litebox_broker_core/src/process.rs @@ -10,6 +10,7 @@ use core::any::Any; use core::ops::Range; use core::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}; +use crate::id::IdAllocator; use crate::object::{self, ObjectEntry, ObjectReference, ObjectRights}; use crate::readiness::{ReadinessRegistration, ReadinessSink}; use crate::signal::ProcessSignals; @@ -17,8 +18,9 @@ use crate::{BrokerCore, BrokerError, Result}; use hashbrown::{HashMap, HashSet}; use litebox_broker_protocol::fs::{FileOpenFlags, FileStatusFlags, SetStatusFlagsRequest}; use litebox_broker_protocol::process::{ - CreatedProcess, ProcessExitStatus, ProcessIdentity, ProcessTermination, + ChildExit, ChildSelector, ProcessExitStatus, ProcessIdentity, ProcessInfo, }; +use litebox_broker_protocol::process_group::ProcessGroupMembership; use litebox_broker_protocol::readiness::ReadinessFlags; use litebox_broker_protocol::{ObjectHandle, ProcessId, ThreadId}; use spin::{Mutex, Once, rwlock::RwLock}; @@ -148,55 +150,6 @@ impl AssociationCancellation { } } -impl ObjectEntry { - fn as_process(&self) -> Result<&ProcessObject> { - match self { - Self::Process(process) => Ok(process), - _ => Err(BrokerError::InvalidRights), - } - } -} - -/// Reference to one process. -/// -/// The reference keeps the process's exit status available until it closes. -pub(crate) struct ProcessObject { - process: Arc, - readiness: ReadinessRegistration, -} - -impl ProcessObject { - pub(crate) fn readiness(&self) -> ReadinessFlags { - if self.process.state.lock().status.is_terminated() { - ReadinessFlags::READ - } else { - ReadinessFlags::default() - } - } - - fn termination(&self) -> Result { - let state = self.process.state.lock(); - let exit_status = match state.status { - ProcessStatus::Starting | ProcessStatus::Running | ProcessStatus::Exiting => { - return Err(BrokerError::WouldBlock); - } - ProcessStatus::Failed(_) => ProcessExitStatus::Unknown, - ProcessStatus::Zombie(status) => status, - }; - Ok(ProcessTermination { - exit_status, - reaped: state.reaped, - }) - } -} - -impl Drop for ProcessObject { - fn drop(&mut self) { - // The child may still hold a registration clone for exit publication. - self.readiness.retire(); - } -} - struct ProcessReferences { handles: Vec, pending_handles: usize, @@ -213,10 +166,13 @@ pub struct BrokerProcess { pub(crate) id: ProcessId, /// ID assigned to the initial thread when process creation completes. initial_thread_id: Once, - /// Creating parent process. - parent: Option>, - /// Whether this process's children are reaped when they terminate. - reap_children: AtomicBool, + /// Process that created this one, if any. + creator: Option, + /// This process's place in the process tree. + /// + /// No other lock is taken while this one is held. Changes are serialized + /// by the core's process tree lock. + links: Mutex, state: Mutex, /// Broker-entry-authenticated caller credential for this process. pub(crate) caller_credential: CallerCredential, @@ -228,21 +184,41 @@ pub struct BrokerProcess { pub(crate) reserved_pipe_capacity: Arc, /// Socket quota held by pending, live, and closing in-flight resources. pub(crate) reserved_sockets: Arc, - /// Signals sent to this process that it has not taken. + /// Signals and child exits this process has not taken. pub(crate) signals: Arc, /// Cancellation state of this process's association. pub(crate) cancellation: AssociationCancellation, } +/// A process's place in the process tree. +struct ProcessLinks { + /// Process group and session. + membership: ProcessGroupMembership, + /// Process that reaps this one, if any. + /// + /// This starts as the creator. Once the parent exits, it becomes the + /// nearest running ancestor that adopts orphans, or none. + parent: Option>, + /// Children not yet reaped, in the order this process gained them. + /// + /// Each keeps its child, and thus a zombie's exit status, alive. + children: Vec>, + /// Whether children are reaped as soon as they exit. + reap_children: bool, + /// Whether this process adopts the orphaned children of its exiting + /// descendants. + adopts_orphans: bool, + /// Exit status, once this process exited. + exit_status: Option, + /// Whether this process no longer exists for other processes, because it + /// was reaped or never started. + released: bool, +} + struct BrokerProcessState { status: ProcessStatus, /// Final termination status reported by the running process. reported_exit_status: Option, - /// Whether this process was reaped when it terminated, so its handles - /// observe termination without a status. - reaped: bool, - /// Parent handle readiness published once this process terminates. - exit_readiness: Option, /// Child retained until this process requests startup. pending_child_process: Option, /// Whether a starting child continues after its parent dies. @@ -276,10 +252,6 @@ pub(crate) enum ProcessStatus { } impl ProcessStatus { - const fn is_terminated(self) -> bool { - matches!(self, Self::Failed(_) | Self::Zombie(_)) - } - fn transition(&mut self, next: Self) -> Result<()> { let allowed = matches!( (*self, next), @@ -295,14 +267,6 @@ impl ProcessStatus { } } -fn publish_exit_readiness(readiness: Option) { - if let Some(readiness) = readiness { - // The association readiness budget reserves capacity for every child - // handle, so publication fails only after the association closes. - let _ = readiness.publish(ReadinessFlags::READ); - } -} - #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum ProcessRetirement { Active { abnormal: bool }, @@ -339,24 +303,41 @@ enum ProcessShutdownRequest { } impl BrokerProcess { - /// Creates authenticated broker process state. + /// Creates authenticated broker process state for the allocated ID `id`. + /// + /// A child of `creator` starts in its group and session, while a root + /// process leads its own. pub(crate) fn new( core: BrokerCore, id: ProcessId, - parent: Option>, + creator: Option<&Arc>, caller_credential: CallerCredential, ) -> Self { + let membership = creator.map_or( + ProcessGroupMembership { + process_group: id, + session: id, + }, + |creator| creator.membership(), + ); + Self::retain_membership_ids(&mut core.ids.lock(), membership); Self { core, id, initial_thread_id: Once::new(), - parent, - reap_children: AtomicBool::new(false), + creator: creator.map(|creator| creator.id), + links: Mutex::new(ProcessLinks { + membership, + parent: creator.map(Arc::downgrade), + children: Vec::new(), + reap_children: false, + adopts_orphans: false, + exit_status: None, + released: false, + }), state: Mutex::new(BrokerProcessState { status: ProcessStatus::Starting, reported_exit_status: None, - reaped: false, - exit_readiness: None, pending_child_process: None, continue_startup_on_parent_death: false, retirement: ProcessRetirement::Active { abnormal: false }, @@ -443,12 +424,9 @@ impl BrokerProcess { /// Allocates and retains one pending child process. /// - /// The returned parent-owned handle publishes readiness through - /// `readiness_sink` once the child terminates. - pub fn allocate_child_process( - &self, - readiness_sink: Arc, - ) -> Result { + /// The child is this process's child from the start, so its exit is + /// reported here even if it exits before starting its own runner. + pub fn allocate_child_process(&self) -> Result { if !self.core.policy.process_duplication_enabled() { return Err(BrokerError::PolicyDenied); } @@ -462,7 +440,9 @@ impl BrokerProcess { } } - let (child, handle) = self.create_child_process(readiness_sink)?; + let child = self + .core + .create_process(self.caller_credential, Some(self.id))?; let result = { let mut state = self.state.lock(); if !self.accepts_operations(&state) { @@ -479,60 +459,24 @@ impl BrokerProcess { process: Arc::clone(&child), image: None, }); - Ok(CreatedProcess { - identity: ProcessIdentity { - process_id: child.id(), - initial_thread_id: child.initial_thread_id(), - }, - handle, + Ok(ProcessIdentity { + process_id: child.id(), + initial_thread_id: child.initial_thread_id(), }) } } }; - if let Err(error) = result { - child.retire(true); - return Err(self - .close_object_reference(handle) - .map_or(BrokerError::Internal, |()| error)); + if result.is_err() { + child.discard(); } result } - /// Creates a child process and the parent-owned handle that observes its - /// termination. - /// - /// The handle publishes readiness through `readiness_sink` once the child - /// terminates, and keeps the child's exit status available until it - /// closes. The child is retired if handle creation fails. - fn create_child_process( - &self, - readiness_sink: Arc, - ) -> Result<(Arc, ObjectHandle)> { - let rights = self - .core - .policy - .principal_object_rights(self.caller_credential)?; - let reference = self.reserve_object_reference(rights)?; - let child = self - .core - .create_process(self.caller_credential, Some(self.id))?; - let readiness = ReadinessRegistration::new_with_retirement_guard( - reference.handle(), - readiness_sink, - Arc::new(()), - ); - child.state.lock().exit_readiness = Some(readiness.clone()); - let object = ProcessObject { - process: Arc::clone(&child), - readiness, - }; - match reference.commit(ObjectEntry::Process(object)) { - Ok(handle) => Ok((child, handle)), - Err(error) => { - child.retire(true); - Err(error) - } - } + /// Discards a child that never started, as if it had never been created, + /// except that its parent learns it was removed. + fn discard(&self) { + let _ = self.fail_start(BrokerError::PeerClosed, false, true); + self.retire(true); } /// Takes the pending child selected for startup. @@ -642,6 +586,14 @@ impl BrokerProcess { child.finish_exit(exit_status, release_thread_ids) } + /// Discards the pending child selected by `child_process_id`, as if it + /// had never been created, except that this process learns it was + /// removed. + pub fn cancel_child_process(&self, child_process_id: ProcessId) -> Result<()> { + self.take_child_process(child_process_id)?.discard(); + Ok(()) + } + /// Records the final termination status reported by this running process. /// /// A runner observes its guest's termination more precisely than its host @@ -656,21 +608,99 @@ impl BrokerProcess { Ok(()) } - /// Sets whether this running process's children are reaped when they - /// terminate. + /// Sets whether this running process's children are reaped as soon as + /// they exit. /// - /// Like Linux `SIGCHLD` auto-reaping, each child applies the setting in - /// effect when it terminates, so children that already terminated keep - /// their status. + /// Each child applies the setting in effect when it exits, so children + /// that already exited are kept until reaped. pub fn set_child_reaping(&self, enabled: bool) -> Result<()> { - let state = self.state.lock(); - if !self.accepts_operations(&state) { + self.update_links(|links| links.reap_children = enabled) + } + + /// Sets whether this running process adopts the orphaned children of its + /// exiting descendants. + /// + /// When a process exits, its children move to its nearest running ancestor + /// that adopts orphans. Without one, they have no parent and are reaped as + /// soon as they exit. + pub fn set_orphan_adoption(&self, enabled: bool) -> Result<()> { + self.update_links(|links| links.adopts_orphans = enabled) + } + + /// Updates this running process's place in the process tree. + fn update_links(&self, update: impl FnOnce(&mut ProcessLinks)) -> Result<()> { + let _tree = self.core.process_tree.lock(); + if !self.accepts_operations(&self.state.lock()) { return Err(BrokerError::PeerClosed); } - self.reap_children.store(enabled, Ordering::SeqCst); + update(&mut self.links.lock()); Ok(()) } + /// Reaps this running process's oldest exited child that `selector` + /// matches, returning its exit. + /// + /// Returns `WouldBlock` if only live children match, and `UnknownObject` + /// if none does. Each matching child's exit or removal is then posted to + /// this process's signals. + pub fn reap_child(&self, selector: ChildSelector) -> Result { + let _tree = self.core.process_tree.lock(); + if !self.accepts_operations(&self.state.lock()) { + return Err(BrokerError::PeerClosed); + } + // This process's links are released while its children's are read, + // since no other lock is taken while one is held. + let mut children = core::mem::take(&mut self.links.lock().children); + let mut matched = false; + let exited = children.iter().position(|child| { + let links = child.links.lock(); + let selected = match selector { + ChildSelector::Any => true, + ChildSelector::Process(id) => child.id == id, + ChildSelector::ProcessGroup(id) => links.membership.process_group == id, + }; + matched |= selected; + selected && links.exit_status.is_some() + }); + let reaped = exited.map(|index| children.remove(index)); + self.links.lock().children = children; + let Some(child) = reaped else { + return Err(if matched { + BrokerError::WouldBlock + } else { + BrokerError::UnknownObject + }); + }; + let mut links = child.links.lock(); + links.released = true; + Ok(ChildExit { + process_id: child.id, + exit_status: links.exit_status.ok_or(BrokerError::Internal)?, + }) + } + + /// Returns the place in the process tree of the process `target`. + /// + /// Returns `UnknownObject` if no such process exists. + pub fn process_info(&self, target: ProcessId) -> Result { + let target = self.core.registered_process(target)?; + // While the links reference a parent, it has yet to finish exiting, + // so it is still registered; once an adoption replaces it, it may be + // reaped. Upgrade it under the links, and drop it only after them. + let (parent, membership) = { + let links = target.links.lock(); + ( + links.parent.as_ref().and_then(Weak::upgrade), + links.membership, + ) + }; + Ok(ProcessInfo { + creator: target.creator, + parent: parent.as_ref().map(|parent| parent.id), + membership, + }) + } + /// Duplicates object references into the pending child selected by /// `child_process_id`, as a Linux child inherits its parent's descriptors. /// @@ -709,22 +739,76 @@ impl BrokerProcess { self.duplicate_object_references_to(handles, child) } - /// Returns this process's parent if it still exists. + /// Returns this process's group and session. + pub(crate) fn membership(&self) -> ProcessGroupMembership { + self.links.lock().membership + } + + /// Moves this process into another group or session. /// - /// Callers upgrade before taking this process's state lock and drop the - /// parent after releasing it, since the last parent reference may run - /// its teardown. - fn live_parent(&self) -> Option> { - self.parent.as_ref().and_then(Weak::upgrade) + /// Callers hold the core's process tree lock. + pub(crate) fn set_membership(&self, membership: ProcessGroupMembership) { + let mut ids = self.core.ids.lock(); + Self::retain_membership_ids(&mut ids, membership); + let previous = core::mem::replace(&mut self.links.lock().membership, membership); + Self::release_membership_ids(&mut ids, previous); + } + + /// Keeps the IDs of the group and session in `membership` from being + /// reused while a process belongs to them, as Linux keeps a PID in use as + /// a process group or session ID. + fn retain_membership_ids(ids: &mut IdAllocator, membership: ProcessGroupMembership) { + ids.retain(membership.process_group.0); + ids.retain(membership.session.0); } - /// Records whether this terminating process is reaped. + /// Releases the IDs that [`Self::retain_membership_ids`] retained. + fn release_membership_ids(ids: &mut IdAllocator, membership: ProcessGroupMembership) { + ids.release(membership.process_group.0); + ids.release(membership.session.0); + } + + /// Returns whether another process created this one. + pub(crate) const fn has_creator(&self) -> bool { + self.creator.is_some() + } + + /// Returns whether this process was reaped or never started, so it no + /// longer exists for other processes. + pub(crate) fn is_released(&self) -> bool { + self.links.lock().released + } + + /// Reserves room for one more child. /// - /// Callers hold this process's state lock across the terminal status - /// transition, so a parent that observes this process live before changing - /// its reaping setting also observes this termination apply that setting. - fn record_reaping(state: &mut BrokerProcessState, parent: Option<&BrokerProcess>) { - state.reaped = parent.is_some_and(|parent| parent.reap_children.load(Ordering::SeqCst)); + /// Callers hold the core's process tree lock. + pub(crate) fn reserve_child(&self) -> Result<()> { + self.links + .lock() + .children + .try_reserve(1) + .map_err(|_| BrokerError::OutOfMemory) + } + + /// Adds a newly created child, for which [`Self::reserve_child`] made + /// room. + /// + /// Callers hold the core's process tree lock. + pub(crate) fn add_child(&self, child: Arc) { + self.links.lock().children.push(child); + } + + /// Calls `f` with the pending child selected by `child_process_id`, which + /// stays pending until `f` returns. + pub(crate) fn with_pending_child( + &self, + child_process_id: ProcessId, + f: impl FnOnce(&Arc) -> R, + ) -> Result { + let mut state = self.state.lock(); + Ok(f(&self + .pending_child(&mut state, child_process_id)? + .process)) } /// Returns whether this process completed broker startup. @@ -773,11 +857,11 @@ impl BrokerProcess { /// Publishes one runner termination outcome. /// /// Exit releases object references, socket state, and, after clean - /// retirement, thread IDs. A zombie keeps only its process ID, registry - /// entry, and exit status, and remains until its parent's process handle - /// and runner supervision release it. The parent's handle becomes - /// readable. Only the first completion releases resources and records its - /// status; later or concurrent completions return without effect. + /// retirement, thread IDs. A zombie keeps only its process ID, group and + /// session, registry entry, and exit status, and remains until its parent reaps it and + /// runner supervision releases it. Only the first completion releases + /// resources and records its status; later or concurrent completions + /// return without effect. /// /// A status reported through [`Self::report_exit_status`] takes precedence /// over `exit_status`. @@ -805,6 +889,10 @@ impl BrokerProcess { /// Releases the resources of a process whose exit this caller claimed, /// then makes it a zombie retaining `exit_status`. + /// + /// The exit is reported to the parent, which keeps the zombie until it + /// reaps it. This process's children move to the nearest of its running + /// ancestors that adopts orphans, if any. fn finish_exit(&self, exit_status: ProcessExitStatus, release_thread_ids: bool) -> Result<()> { // Like Linux, release resources before the exit becomes observable, so // a parent that reaps the child sees its pipes and sockets closed. @@ -813,21 +901,142 @@ impl BrokerProcess { } else if release_thread_ids { self.release_threads(true); } - let parent = self.live_parent(); - let exit_readiness = { - let mut state = self.state.lock(); - state + // Children that cannot start without this process fail, as when its + // owner dies. + self.handle_owner_death(); + { + let _tree = self.core.process_tree.lock(); + self.state + .lock() .status .transition(ProcessStatus::Zombie(exit_status))?; - Self::record_reaping(&mut state, parent.as_deref()); - state.exit_readiness.take() - }; - drop(parent); + let (parent, children) = { + let mut links = self.links.lock(); + links.exit_status = Some(exit_status); + ( + links.parent.as_ref().and_then(Weak::upgrade), + core::mem::take(&mut links.children), + ) + }; + let adopter = Self::nearest_adopter(parent.clone()).filter(|adopter| { + // Without room for the orphans, they are left without a + // parent. + adopter + .links + .lock() + .children + .try_reserve(children.len()) + .is_ok() + }); + for child in children { + Self::adopt(adopter.as_ref(), child); + } + self.report_exit(parent.as_deref(), exit_status); + } self.core.process_lifecycle_sink.changed(); - publish_exit_readiness(exit_readiness); Ok(()) } + /// Returns the nearest of `ancestor` and its ancestors that is running and + /// adopts orphans. + /// + /// Like Linux, this skips exiting ancestors, which could otherwise reap a + /// zombie orphan before passing their children on. + /// Callers hold the core's process tree lock. + fn nearest_adopter(mut ancestor: Option>) -> Option> { + while let Some(process) = ancestor { + let running = process.accepts_operations(&process.state.lock()); + let links = process.links.lock(); + if running && links.adopts_orphans { + drop(links); + return Some(process); + } + let parent = links.parent.as_ref().and_then(Weak::upgrade); + drop(links); + ancestor = parent; + } + None + } + + /// Makes `parent`, which has room for it, the parent of the orphan + /// `child`, or leaves `child` without one. + /// + /// A zombie reports its exit to its new parent as if it just exited. + /// Callers hold the core's process tree lock. + fn adopt(parent: Option<&Arc>, child: Arc) { + let exit_status = { + let mut links = child.links.lock(); + links.parent = parent.map(Arc::downgrade); + links.exit_status + }; + if let Some(parent) = parent { + parent.links.lock().children.push(Arc::clone(&child)); + } + if let Some(exit_status) = exit_status { + child.report_exit(parent.map(|parent| &**parent), exit_status); + } + } + + /// Reports this zombie's exit to `parent`, which keeps this process + /// unless it reaps children as soon as they exit. Without a parent, this + /// process is reaped at once. + /// + /// Callers hold the core's process tree lock. + fn report_exit(&self, parent: Option<&Self>, exit_status: ProcessExitStatus) { + let mut reaped = parent.is_none(); + // The last reference may run this process's teardown, so it drops + // only after the parent's links are released. + let mut removed = None; + if let Some(parent) = parent { + { + let mut links = parent.links.lock(); + if links.reap_children { + reaped = true; + let index = links + .children + .iter() + .position(|child| core::ptr::eq(Arc::as_ptr(child), self)); + removed = index.map(|index| links.children.remove(index)); + } + } + parent.signals.post_child_exit(ChildExit { + process_id: self.id, + exit_status, + }); + } + if reaped { + self.links.lock().released = true; + } + drop(removed); + } + + /// Removes this process, whose startup failed, from the process tree, + /// telling its parent, whose reaps may have found it live. + /// + /// Callers hold the core's process tree lock. + fn release_unstarted(&self) { + let parent = { + let mut links = self.links.lock(); + links.released = true; + links.parent.take().as_ref().and_then(Weak::upgrade) + }; + let Some(parent) = parent else { + return; + }; + let removed = { + let mut links = parent.links.lock(); + let index = links + .children + .iter() + .position(|child| core::ptr::eq(Arc::as_ptr(child), self)); + index.map(|index| links.children.remove(index)) + }; + if removed.is_some() { + parent.signals.post_child_removed(); + } + drop(removed); + } + /// Installs the host runner termination action. pub fn install_shutdown(&self, shutdown: ProcessShutdown) { let shutdown = { @@ -841,44 +1050,48 @@ impl BrokerProcess { } /// Fails pending startup and returns the authoritative startup outcome. + /// + /// A process whose startup failed leaves the process tree, as if it had + /// never been created, except that its parent learns it was removed. pub fn fail_start( &self, error: BrokerError, abnormal: bool, expected_shutdown: bool, ) -> Result<()> { - let parent = self.live_parent(); - let (shutdown, exit_readiness) = { - let mut state = self.state.lock(); - match state.status { - ProcessStatus::Starting => {} - ProcessStatus::Running | ProcessStatus::Exiting | ProcessStatus::Zombie(_) => { - return Ok(()); + let shutdown = { + let _tree = self.core.process_tree.lock(); + let shutdown = { + let mut state = self.state.lock(); + match state.status { + ProcessStatus::Starting => {} + ProcessStatus::Running | ProcessStatus::Exiting | ProcessStatus::Zombie(_) => { + return Ok(()); + } + ProcessStatus::Failed(error) => return Err(error), } - ProcessStatus::Failed(error) => return Err(error), - } - if abnormal { - state.retirement.mark_abnormal(); - } - state.status.transition(ProcessStatus::Failed(error))?; - Self::record_reaping(&mut state, parent.as_deref()); - state.continue_startup_on_parent_death = false; - if state.shutdown_request == ProcessShutdownRequest::None { - state.shutdown_request = if expected_shutdown { - ProcessShutdownRequest::Expected - } else { - ProcessShutdownRequest::Unexpected - }; - } - (state.shutdown.clone(), state.exit_readiness.take()) + if abnormal { + state.retirement.mark_abnormal(); + } + state.status.transition(ProcessStatus::Failed(error))?; + state.continue_startup_on_parent_death = false; + if state.shutdown_request == ProcessShutdownRequest::None { + state.shutdown_request = if expected_shutdown { + ProcessShutdownRequest::Expected + } else { + ProcessShutdownRequest::Unexpected + }; + } + state.shutdown.clone() + }; + self.release_unstarted(); + shutdown }; - drop(parent); // Release the child's references before publishing the failure. if self.release_references() { self.state.lock().retirement.mark_abnormal(); } self.core.process_lifecycle_sink.changed(); - publish_exit_readiness(exit_readiness); if let Some(shutdown) = shutdown { shutdown(); } @@ -901,33 +1114,27 @@ impl BrokerProcess { /// Applies owner-death handling to every direct child process. /// /// Ordinary startup fails and prepared duplication startup continues. - /// Children become orphans; their exit status remains only while this - /// process still holds their handles. + /// Calling this method more than once is harmless. pub fn handle_owner_death(&self) { self.cancellation.cancel(); - // Child creation checks cancellation under the state lock, so once this - // lock is taken every child created for a live owner is visible below. let pending_child_process = self.state.lock().pending_child_process.take(); - // Drop non-children only after releasing the registry lock, because a - // final process drop removes itself from the registry. - let processes = { - let processes = self.core.processes.read(); - processes - .values() - .filter_map(Weak::upgrade) - .collect::>() - }; - - let mut changed = false; let mut shutdowns = Vec::new(); - for child in processes.iter().filter(|process| process.is_child_of(self)) { - let (child_changed, shutdown) = child.handle_parent_death(); - changed |= child_changed; - if let Some(shutdown) = shutdown { - shutdowns.push(shutdown); - } - } - drop(processes); + let changed = { + // Child creation checks cancellation under the tree lock, so once + // this lock is taken every child created for a live owner is + // listed. + let _tree = self.core.process_tree.lock(); + let mut children = core::mem::take(&mut self.links.lock().children); + let count = children.len(); + children.retain(|child| { + let (failed, shutdown) = child.handle_parent_death(); + shutdowns.extend(shutdown); + !failed + }); + let changed = children.len() != count; + self.links.lock().children = children; + changed + }; if changed { self.core.process_lifecycle_sink.changed(); } @@ -935,8 +1142,7 @@ impl BrokerProcess { shutdown(); } if let Some(PendingChild { process: child, .. }) = pending_child_process { - let _ = child.fail_start(BrokerError::PeerClosed, false, true); - child.retire(true); + child.discard(); } } @@ -969,15 +1175,21 @@ impl BrokerProcess { self.is_active(state) && matches!(state.status, ProcessStatus::Starting) } - fn is_child_of(&self, parent: &BrokerProcess) -> bool { + pub(crate) fn is_child_of(&self, parent: &BrokerProcess) -> bool { // The weak reference keeps the parent's allocation, so its address // cannot be reused by another process. - self.parent + self.links + .lock() + .parent .as_ref() .is_some_and(|own_parent| core::ptr::eq(own_parent.as_ptr(), parent)) } - /// Fails ordinary startup after the parent's owner dies. + /// Fails ordinary startup after the parent's owner dies, returning + /// whether it failed and the runner shutdown to run. + /// + /// A child whose startup failed leaves the process tree. Callers hold the + /// core's process tree lock and remove the child from its parent. fn handle_parent_death(&self) -> (bool, Option) { let mut state = self.state.lock(); if state.status != ProcessStatus::Starting || state.continue_startup_on_parent_death { @@ -990,6 +1202,9 @@ impl BrokerProcess { if state.shutdown_request == ProcessShutdownRequest::None { state.shutdown_request = ProcessShutdownRequest::Expected; } + let mut links = self.links.lock(); + links.released = true; + links.parent = None; (true, state.shutdown.clone()) } @@ -1448,16 +1663,6 @@ impl BrokerProcess { object::set_status_flags(self, &object, request.mask, request.flags) } - /// Returns a process's termination status through a process handle. - /// - /// Returns `WouldBlock` while the process is live. A process whose startup - /// failed reports [`ProcessExitStatus::Unknown`]. - pub fn process_exit_status(&self, handle: ObjectHandle) -> Result { - let object = self.authorized_object(handle, ObjectRights::WAIT)?; - let object = object.read(); - object.as_process()?.termination() - } - /// Closes one object reference owned by this process. /// /// The underlying object is released when this was the last live reference. @@ -1525,7 +1730,9 @@ impl BrokerProcess { /// /// Set `release_ids` only after fully accounted teardown. Final process /// drop otherwise retains IDs after an unwind or uncertain retirement. - /// Calling this method more than once is harmless. + /// Calling this method more than once is harmless. It never takes the + /// core's process tree lock, so a process may drop while that lock is + /// held. pub fn cleanup(&self, release_ids: bool) { let release_ids = { let mut state = self.state.lock(); @@ -1547,7 +1754,10 @@ impl BrokerProcess { let release_ids = release_ids && !invariant_fault; self.release_threads(release_ids); if release_ids { - self.core.ids.lock().release(self.id.0); + let membership = self.membership(); + let mut ids = self.core.ids.lock(); + ids.release(self.id.0); + Self::release_membership_ids(&mut ids, membership); } self.core.process_lifecycle_sink.changed(); } @@ -1747,8 +1957,9 @@ mod tests { use litebox_broker_protocol::fs::{ FileAccessMode, FileError, FileMode, FileOpenFlags, FileSeekWhence, FileType, FileUser, }; - use litebox_broker_protocol::process::{CreatedProcess, ProcessExitStatus, ProcessTermination}; + use litebox_broker_protocol::process::{ChildExit, ChildSelector, ProcessExitStatus}; use litebox_broker_protocol::readiness::ReadinessFlags; + use litebox_broker_protocol::signal::SignalEvent; use litebox_broker_protocol::{ObjectHandle, ProcessId}; use std::{boxed::Box, sync::Arc, vec, vec::Vec}; @@ -1760,13 +1971,46 @@ mod tests { const EXITED: ProcessExitStatus = ProcessExitStatus::Exited { code: 23 }; const SIGNALED: ProcessExitStatus = ProcessExitStatus::Signaled { signal: 11 }; - fn termination(exit_status: ProcessExitStatus, reaped: bool) -> ProcessTermination { - ProcessTermination { + fn child_exit(child: &BrokerProcess, exit_status: ProcessExitStatus) -> ChildExit { + ChildExit { + process_id: child.id(), exit_status, - reaped, } } + fn child_exited(child: &BrokerProcess, exit_status: ProcessExitStatus) -> SignalEvent { + SignalEvent::ChildExited(child_exit(child, exit_status)) + } + + /// Creates a starting child of `parent`. + fn child_of(parent: &BrokerProcess) -> Arc { + parent + .core + .create_process(parent.caller_credential(), Some(parent.id())) + .unwrap() + } + + /// Starts `process` if it is starting, then has it exit with + /// `exit_status`. + fn run_to_exit(process: &BrokerProcess, exit_status: ProcessExitStatus) { + if process.startup_result().is_none() { + process.complete_start().unwrap(); + } + process.retire(true); + process.complete_exit(exit_status).unwrap(); + } + + /// Opens `process`'s signals, through which it learns of child exits. + fn open_signals( + process: &BrokerProcess, + ) -> ( + ObjectHandle, + Arc, + ) { + let sink = readiness_sink(); + (crate::signal::open(process, sink.clone()).unwrap(), sink) + } + #[derive(Default)] struct TestProcessLifecycleSink { changes: AtomicUsize, @@ -1780,6 +2024,8 @@ mod tests { fn parent_id(process: &BrokerProcess) -> Option { process + .links + .lock() .parent .as_ref() .and_then(alloc::sync::Weak::upgrade) @@ -1862,8 +2108,7 @@ mod tests { let parent_sink = readiness_sink(); let child_sink = readiness_sink(); let (reader, writer) = crate::pipe::create(&parent, 4, 2, FileOpenFlags::NONE).unwrap(); - let CreatedProcess { identity, .. } = - parent.allocate_child_process(parent_sink.clone()).unwrap(); + let identity = parent.allocate_child_process().unwrap(); // Sharing only the write end still lets the child wake the parent's reader. let child_writer = parent .duplicate_object_references_to_child( @@ -1924,7 +2169,7 @@ mod tests { } #[test] - fn child_handle_retains_zombie_until_closed() { + fn zombie_remains_until_reaped() { let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( ObjectRights::all(), )) @@ -1935,23 +2180,19 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); root.complete_start().unwrap(); - let sink = readiness_sink(); - let (child, handle) = root.create_child_process(sink.clone()).unwrap(); + let (signals, sink) = open_signals(&root); + let child = child_of(&root); let child_id = child.id(); assert_eq!( - root.process_exit_status(handle), + root.reap_child(ChildSelector::Any), Err(BrokerError::WouldBlock) ); - assert_eq!(root.check_readiness(handle), Ok(ReadinessFlags::default())); + assert_eq!(root.check_readiness(signals), Ok(ReadinessFlags::default())); child.complete_start().unwrap(); let (read, write) = crate::pipe::create(&root, 1, 1, FileOpenFlags::NONE).unwrap(); root.duplicate_object_reference_to(write, &child, ObjectRights::WRITE) .unwrap(); root.close_object_reference(write).unwrap(); - assert_eq!( - root.duplicate_object_reference_to(handle, &child, ObjectRights::WAIT), - Err(BrokerError::UnsupportedOperation) - ); child.retire(true); child.complete_exit(EXITED).unwrap(); @@ -1959,13 +2200,15 @@ mod tests { drop(child); assert_eq!( - *sink.published.lock().unwrap(), - [(handle, ReadinessFlags::READ)] + *sink.republished.lock().unwrap(), + [(signals, ReadinessFlags::READ)] ); - assert_eq!(root.check_readiness(handle), Ok(ReadinessFlags::READ)); assert_eq!( - root.process_exit_status(handle), - Ok(termination(EXITED, false)) + crate::signal::take(&root, signals), + Ok(SignalEvent::ChildExited(ChildExit { + process_id: child_id, + exit_status: EXITED, + })) ); assert!( root.check_readiness(read) @@ -1978,11 +2221,25 @@ mod tests { broker.allocate_process(CallerCredential::Unauthenticated, None), Err(BrokerError::ResourceExhausted) )); + assert_eq!( + root.process_info(child_id).map(|info| info.parent), + Ok(Some(root.id())) + ); - root.close_object_reference(handle).unwrap(); + assert_eq!( + root.reap_child(ChildSelector::Process(child_id)), + Ok(ChildExit { + process_id: child_id, + exit_status: EXITED, + }) + ); - assert_eq!(*sink.retired.lock().unwrap(), [handle]); assert!(!broker.processes.read().contains_key(&child_id)); + assert_eq!(root.process_info(child_id), Err(BrokerError::UnknownObject)); + assert_eq!( + root.reap_child(ChildSelector::Any), + Err(BrokerError::UnknownObject) + ); assert!( broker .allocate_process(CallerCredential::Unauthenticated, None) @@ -1990,6 +2247,222 @@ mod tests { ); } + #[test] + fn reaping_selects_the_oldest_exited_matching_child() { + let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( + ObjectRights::all(), + )) + .build() + .unwrap(); + let root = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + root.complete_start().unwrap(); + let first = child_of(&root); + let second = child_of(&root); + let third = child_of(&root); + let pending = child_of(&root); + crate::process_group::set(&root, third.id(), third.id()).unwrap(); + let other = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + let reap = |selector| root.reap_child(selector); + + // A child that has not started counts as live. + assert_eq!(reap(ChildSelector::Any), Err(BrokerError::WouldBlock)); + run_to_exit(&third, SIGNALED); + run_to_exit(&second, EXITED); + run_to_exit(&first, EXITED); + assert_eq!( + reap(ChildSelector::Process(other.id())), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + reap(ChildSelector::ProcessGroup(other.id())), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + reap(ChildSelector::ProcessGroup(third.id())), + Ok(child_exit(&third, SIGNALED)) + ); + assert_eq!( + reap(ChildSelector::ProcessGroup(third.id())), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + reap(ChildSelector::Process(second.id())), + Ok(child_exit(&second, EXITED)) + ); + assert_eq!( + reap(ChildSelector::ProcessGroup(root.id())), + Ok(child_exit(&first, EXITED)) + ); + assert_eq!( + reap(ChildSelector::Process(pending.id())), + Err(BrokerError::WouldBlock) + ); + assert_eq!(reap(ChildSelector::Any), Err(BrokerError::WouldBlock)); + + // Only a running process reaps. + assert_eq!( + pending.reap_child(ChildSelector::Any), + Err(BrokerError::PeerClosed) + ); + run_to_exit(&pending, EXITED); + assert_eq!(reap(ChildSelector::Any), Ok(child_exit(&pending, EXITED))); + assert_eq!(reap(ChildSelector::Any), Err(BrokerError::UnknownObject)); + } + + #[test] + fn orphans_move_to_the_nearest_adopting_ancestor() { + let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( + ObjectRights::all(), + )) + .build() + .unwrap(); + let root = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + root.complete_start().unwrap(); + let adopter = child_of(&root); + adopter.complete_start().unwrap(); + let middle = child_of(&adopter); + middle.complete_start().unwrap(); + let exiting = child_of(&middle); + exiting.complete_start().unwrap(); + let running = child_of(&exiting); + running.complete_start().unwrap(); + let zombie = child_of(&exiting); + run_to_exit(&zombie, SIGNALED); + root.set_orphan_adoption(true).unwrap(); + adopter.set_orphan_adoption(true).unwrap(); + let (signals, _) = open_signals(&adopter); + + run_to_exit(&exiting, EXITED); + + // The adopter learns of the adopted zombie's exit, and reaps it after + // its own children. + assert_eq!(parent_id(&running), Some(adopter.id())); + assert_eq!(parent_id(&zombie), Some(adopter.id())); + assert_eq!( + crate::signal::take(&adopter, signals), + Ok(child_exited(&zombie, SIGNALED)) + ); + assert_eq!( + middle.reap_child(ChildSelector::Any), + Ok(child_exit(&exiting, EXITED)) + ); + assert_eq!( + adopter.reap_child(ChildSelector::Any), + Ok(child_exit(&zombie, SIGNALED)) + ); + assert_eq!( + adopter.reap_child(ChildSelector::Any), + Err(BrokerError::WouldBlock) + ); + let info = root.process_info(running.id()).unwrap(); + assert_eq!(info.creator, Some(exiting.id())); + assert_eq!(info.parent, Some(adopter.id())); + + // Once the adopter exits, the next adopting ancestor adopts. + run_to_exit(&adopter, EXITED); + assert_eq!(parent_id(&running), Some(root.id())); + assert_eq!(parent_id(&middle), Some(root.id())); + run_to_exit(&running, EXITED); + assert_eq!( + root.reap_child(ChildSelector::Any), + Ok(child_exit(&adopter, EXITED)) + ); + assert_eq!( + root.reap_child(ChildSelector::Any), + Ok(child_exit(&running, EXITED)) + ); + assert_eq!( + root.reap_child(ChildSelector::Any), + Err(BrokerError::WouldBlock) + ); + } + + #[test] + fn orphans_skip_exiting_adopters() { + let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( + ObjectRights::all(), + )) + .build() + .unwrap(); + let root = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + root.complete_start().unwrap(); + let adopter = child_of(&root); + adopter.complete_start().unwrap(); + let middle = child_of(&adopter); + middle.complete_start().unwrap(); + let exiting = child_of(&middle); + exiting.complete_start().unwrap(); + let running = child_of(&exiting); + running.complete_start().unwrap(); + let zombie = child_of(&exiting); + run_to_exit(&zombie, SIGNALED); + adopter.set_orphan_adoption(true).unwrap(); + middle.set_orphan_adoption(true).unwrap(); + middle.set_child_reaping(true).unwrap(); + let (signals, _) = open_signals(&adopter); + // Another caller has claimed the middle process's exit. + middle.state.lock().status = ProcessStatus::Exiting; + + run_to_exit(&exiting, EXITED); + + // The exiting middle process would have reaped the zombie at once. + assert_eq!(parent_id(&running), Some(adopter.id())); + assert_eq!(parent_id(&zombie), Some(adopter.id())); + assert_eq!( + crate::signal::take(&adopter, signals), + Ok(child_exited(&zombie, SIGNALED)) + ); + assert_eq!( + adopter.reap_child(ChildSelector::Any), + Ok(child_exit(&zombie, SIGNALED)) + ); + } + + #[test] + fn orphans_without_an_adopter_are_reaped_when_they_exit() { + let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( + ObjectRights::all(), + )) + .build() + .unwrap(); + let root = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + root.complete_start().unwrap(); + let zombie = child_of(&root); + let zombie_id = zombie.id(); + run_to_exit(&zombie, EXITED); + drop(zombie); + let running = child_of(&root); + let running_id = running.id(); + running.complete_start().unwrap(); + + root.handle_owner_death(); + root.complete_exit(EXITED).unwrap(); + + // The exited root and its zombie child no longer exist. + assert!(!broker.processes.read().contains_key(&zombie_id)); + assert_eq!( + running.process_info(root.id()), + Err(BrokerError::UnknownObject) + ); + let info = running.process_info(running_id).unwrap(); + assert_eq!(info.creator, Some(root.id())); + assert_eq!(info.parent, None); + running.retire(true); + running.complete_exit(EXITED).unwrap(); + drop(running); + assert!(!broker.processes.read().contains_key(&running_id)); + } + #[test] fn exit_in_progress_is_neither_observable_nor_repeated() { let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( @@ -2001,8 +2474,8 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); root.complete_start().unwrap(); - let sink = readiness_sink(); - let (child, handle) = root.create_child_process(sink.clone()).unwrap(); + let (signals, sink) = open_signals(&root); + let child = child_of(&root); child.complete_start().unwrap(); let (read, write) = crate::pipe::create(&root, 1, 1, FileOpenFlags::NONE).unwrap(); root.duplicate_object_reference_to(write, &child, ObjectRights::WRITE) @@ -2021,11 +2494,11 @@ mod tests { .contains(ReadinessFlags::HANGUP) ); assert_eq!( - root.process_exit_status(handle), + root.reap_child(ChildSelector::Any), Err(BrokerError::WouldBlock) ); - assert_eq!(root.check_readiness(handle), Ok(ReadinessFlags::default())); - assert!(sink.published.lock().unwrap().is_empty()); + assert_eq!(root.check_readiness(signals), Ok(ReadinessFlags::default())); + assert!(sink.republished.lock().unwrap().is_empty()); } #[test] @@ -2039,7 +2512,7 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); root.complete_start().unwrap(); - let (child, handle) = root.create_child_process(readiness_sink()).unwrap(); + let child = child_of(&root); assert_eq!( child.report_exit_status(SIGNALED), Err(BrokerError::PeerClosed) @@ -2051,8 +2524,8 @@ mod tests { child.complete_exit(EXITED).unwrap(); assert_eq!( - root.process_exit_status(handle), - Ok(termination(SIGNALED, false)) + root.reap_child(ChildSelector::Any), + Ok(child_exit(&child, SIGNALED)) ); } @@ -2067,81 +2540,55 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); root.complete_start().unwrap(); - let sink = readiness_sink(); - let (zombie, zombie_handle) = root.create_child_process(sink.clone()).unwrap(); - let (reaped, reaped_handle) = root.create_child_process(sink.clone()).unwrap(); - let (failed, failed_handle) = root.create_child_process(sink.clone()).unwrap(); + let (signals, sink) = open_signals(&root); + let zombie = child_of(&root); + let reaped = child_of(&root); + let failed = child_of(&root); assert_eq!(zombie.set_child_reaping(true), Err(BrokerError::PeerClosed)); - let exit = |child: &BrokerProcess| { - child.complete_start().unwrap(); - child.retire(true); - child.complete_exit(EXITED).unwrap(); - }; - exit(&zombie); + run_to_exit(&zombie, EXITED); root.set_child_reaping(true).unwrap(); - exit(&reaped); + run_to_exit(&reaped, SIGNALED); assert_eq!( failed.fail_start(BrokerError::PeerClosed, false, true), Err(BrokerError::PeerClosed) ); + // Both exits are notified, coalescing into the first, but only the + // one before reaping was enabled is kept. A failed child is removed. assert_eq!( - root.process_exit_status(zombie_handle), - Ok(termination(EXITED, false)) + *sink.republished.lock().unwrap(), + [ + (signals, ReadinessFlags::READ), + (signals, ReadinessFlags::READ) + ] ); assert_eq!( - root.process_exit_status(reaped_handle), - Ok(termination(EXITED, true)) + crate::signal::take(&root, signals), + Ok(child_exited(&zombie, EXITED)) ); assert_eq!( - root.process_exit_status(failed_handle), - Ok(termination(ProcessExitStatus::Unknown, true)) + crate::signal::take(&root, signals), + Ok(SignalEvent::ChildRemoved) ); assert_eq!( - *sink.published.lock().unwrap(), - [ - (zombie_handle, ReadinessFlags::READ), - (reaped_handle, ReadinessFlags::READ), - (failed_handle, ReadinessFlags::READ), - ] + crate::signal::take(&root, signals), + Err(BrokerError::WouldBlock) ); - } - - #[test] - fn parent_exit_releases_child_handles() { - let broker = TestBrokerCoreBuilder::new(PolicyEngine::with_unauthenticated_rights( - ObjectRights::all(), - )) - .build() - .unwrap(); - let root = broker - .allocate_process(CallerCredential::Unauthenticated, None) - .unwrap(); - root.complete_start().unwrap(); - let sink = readiness_sink(); - let (zombie, zombie_handle) = root.create_child_process(sink.clone()).unwrap(); - let zombie_id = zombie.id(); - zombie.complete_start().unwrap(); - zombie.retire(true); - zombie.complete_exit(EXITED).unwrap(); - drop(zombie); - let (running, running_handle) = root.create_child_process(sink.clone()).unwrap(); - let running_id = running.id(); - running.complete_start().unwrap(); - - root.handle_owner_death(); - root.complete_exit(EXITED).unwrap(); - + for child in [&reaped, &failed] { + assert_eq!( + root.process_info(child.id()), + Err(BrokerError::UnknownObject) + ); + } assert_eq!( - *sink.retired.lock().unwrap(), - [running_handle, zombie_handle] + root.reap_child(ChildSelector::Any), + Ok(child_exit(&zombie, EXITED)) + ); + assert_eq!( + root.reap_child(ChildSelector::Any), + Err(BrokerError::UnknownObject) ); - assert!(!broker.processes.read().contains_key(&zombie_id)); - running.retire(true); - running.complete_exit(EXITED).unwrap(); - drop(running); - assert!(!broker.processes.read().contains_key(&running_id)); } #[test] @@ -2155,8 +2602,8 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); root.complete_start().unwrap(); - let sink = readiness_sink(); - let (child, handle) = root.create_child_process(sink.clone()).unwrap(); + let (signals, sink) = open_signals(&root); + let child = child_of(&root); let child_id = child.id(); assert_eq!( @@ -2168,13 +2615,23 @@ mod tests { child.state.lock().status, ProcessStatus::Failed(BrokerError::PeerClosed) ); + // The parent learns the child is gone, but not that it exited. + assert_eq!(sink.republished.lock().unwrap().len(), 1); + assert_eq!( + crate::signal::take(&root, signals), + Ok(SignalEvent::ChildRemoved) + ); assert_eq!( - *sink.published.lock().unwrap(), - [(handle, ReadinessFlags::READ)] + crate::signal::take(&root, signals), + Err(BrokerError::WouldBlock) + ); + assert_eq!( + root.reap_child(ChildSelector::Any), + Err(BrokerError::UnknownObject) ); + assert_eq!(root.process_info(child_id), Err(BrokerError::UnknownObject)); child.retire(true); drop(child); - root.close_object_reference(handle).unwrap(); assert!(!broker.processes.read().contains_key(&child_id)); } @@ -2213,7 +2670,7 @@ mod tests { .unwrap(); parent.complete_start().unwrap(); assert_eq!( - parent.allocate_child_process(readiness_sink()), + parent.allocate_child_process(), Err(BrokerError::PolicyDenied) ); } @@ -2231,15 +2688,13 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let sink = readiness_sink(); - let CreatedProcess { identity, handle } = - parent.allocate_child_process(sink.clone()).unwrap(); + let identity = parent.allocate_child_process().unwrap(); let process_id = identity.process_id; assert_eq!( - parent.allocate_child_process(sink.clone()), + parent.allocate_child_process(), Err(BrokerError::WouldBlock) ); - assert_eq!(parent.references.lock().handles, [handle]); + assert_eq!(parent.references.lock().handles, []); assert!(matches!( parent.take_child_process(ProcessId(process_id.0 + 1)), Err(BrokerError::UnknownObject) @@ -2254,6 +2709,103 @@ mod tests { assert!(child.is_running()); } + #[test] + fn cancelled_child_process_vanishes() { + let broker = TestBrokerCoreBuilder::new( + PolicyEngine::with_unauthenticated_rights(ObjectRights::all()) + .with_process_duplication_enabled(true), + ) + .build() + .unwrap(); + let parent = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + parent.complete_start().unwrap(); + let (signals, _) = open_signals(&parent); + let process_id = parent.allocate_child_process().unwrap().process_id; + let child = parent.with_pending_child(process_id, Arc::clone).unwrap(); + assert_eq!( + parent.cancel_child_process(ProcessId(process_id.0 + 1)), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + parent.reap_child(ChildSelector::Any), + Err(BrokerError::WouldBlock) + ); + + parent.cancel_child_process(process_id).unwrap(); + + // A reap that found the child live learns it is gone. + assert_eq!( + crate::signal::take(&parent, signals), + Ok(SignalEvent::ChildRemoved) + ); + assert_eq!(child.startup_result(), Some(Err(BrokerError::PeerClosed))); + assert_eq!( + parent.process_info(process_id), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + parent.reap_child(ChildSelector::Any), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + parent.cancel_child_process(process_id), + Err(BrokerError::UnknownObject) + ); + drop(child); + assert!(!broker.processes.read().contains_key(&process_id)); + // The pending slot is free again. + parent.allocate_child_process().unwrap(); + } + + #[test] + fn adopted_child_that_fails_to_start_is_reported_removed() { + let broker = TestBrokerCoreBuilder::new( + PolicyEngine::with_unauthenticated_rights(ObjectRights::all()) + .with_process_duplication_enabled(true), + ) + .build() + .unwrap(); + let root = broker + .allocate_process(CallerCredential::Unauthenticated, None) + .unwrap(); + root.complete_start().unwrap(); + root.set_orphan_adoption(true).unwrap(); + let parent = child_of(&root); + parent.complete_start().unwrap(); + let starting = prepared_duplication(&parent); + run_to_exit(&parent, EXITED); + assert_eq!(parent_id(&starting), Some(root.id())); + assert_eq!( + root.reap_child(ChildSelector::Any), + Ok(child_exit(&parent, EXITED)) + ); + assert_eq!( + root.reap_child(ChildSelector::Any), + Err(BrokerError::WouldBlock) + ); + let (signals, _) = open_signals(&root); + assert_eq!( + crate::signal::take(&root, signals), + Ok(child_exited(&parent, EXITED)) + ); + + assert_eq!( + starting.fail_start(BrokerError::PeerClosed, false, true), + Err(BrokerError::PeerClosed) + ); + + assert_eq!( + crate::signal::take(&root, signals), + Ok(SignalEvent::ChildRemoved) + ); + assert_eq!( + root.reap_child(ChildSelector::Any), + Err(BrokerError::UnknownObject) + ); + } + #[test] fn owner_death_fails_and_retires_pending_child_process() { let broker = TestBrokerCoreBuilder::new( @@ -2266,9 +2818,7 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let CreatedProcess { identity, handle } = - parent.allocate_child_process(readiness_sink()).unwrap(); - let process_id = identity.process_id; + let process_id = parent.allocate_child_process().unwrap().process_id; let child = broker .processes .read() @@ -2279,10 +2829,8 @@ mod tests { parent.handle_owner_death(); assert_eq!(child.startup_result(), Some(Err(BrokerError::PeerClosed))); - assert_eq!( - parent.process_exit_status(handle), - Ok(termination(ProcessExitStatus::Unknown, false)) - ); + assert!(child.is_released()); + assert!(parent.links.lock().children.is_empty()); assert!(matches!( parent.take_child_process(process_id), Err(BrokerError::PeerClosed) @@ -2303,23 +2851,25 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let sink = readiness_sink(); - let CreatedProcess { identity, handle } = - parent.allocate_child_process(sink.clone()).unwrap(); - let process_id = identity.process_id; + let (signals, sink) = open_signals(&parent); + let process_id = parent.allocate_child_process().unwrap().process_id; let changes = lifecycle.changes.load(Ordering::Relaxed); let threads = broker.active_thread_count.load(Ordering::Relaxed); assert_eq!(parent.exit_child_process(process_id, EXITED), Ok(())); + let exit = ChildExit { + process_id, + exit_status: EXITED, + }; assert_eq!( - parent.process_exit_status(handle), - Ok(termination(EXITED, false)) + crate::signal::take(&parent, signals), + Ok(SignalEvent::ChildExited(exit)) ); assert_eq!(lifecycle.changes.load(Ordering::Relaxed), changes + 1); assert_eq!( - *sink.published.lock().unwrap(), - [(handle, ReadinessFlags::READ)] + *sink.republished.lock().unwrap(), + [(signals, ReadinessFlags::READ)] ); assert_eq!( broker.active_thread_count.load(Ordering::Relaxed), @@ -2330,9 +2880,9 @@ mod tests { Err(BrokerError::UnknownObject) ); - // The zombie stays until its handle closes. + // The zombie stays until reaped. assert!(broker.processes.read().contains_key(&process_id)); - parent.close_object_reference(handle).unwrap(); + assert_eq!(parent.reap_child(ChildSelector::Any), Ok(exit)); assert!(!broker.processes.read().contains_key(&process_id)); } @@ -2363,11 +2913,7 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let process_id = parent - .allocate_child_process(readiness_sink()) - .unwrap() - .identity - .process_id; + let process_id = parent.allocate_child_process().unwrap().process_id; let writes = Arc::new(std::sync::Mutex::new(Vec::new())); let created = AtomicUsize::new(0); let create = |capacity| { @@ -2437,11 +2983,7 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let child = parent - .allocate_child_process(readiness_sink()) - .unwrap() - .identity - .process_id; + let child = parent.allocate_child_process().unwrap().process_id; (parent, child) }; let write = |parent: &BrokerProcess, child, offset, length| { @@ -2495,8 +3037,7 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let CreatedProcess { identity, handle } = - parent.allocate_child_process(readiness_sink()).unwrap(); + let identity = parent.allocate_child_process().unwrap(); parent.handle_owner_death(); @@ -2504,10 +3045,7 @@ mod tests { parent.exit_child_process(identity.process_id, EXITED), Err(BrokerError::PeerClosed) ); - assert_eq!( - parent.process_exit_status(handle), - Ok(termination(ProcessExitStatus::Unknown, false)) - ); + assert!(parent.links.lock().children.is_empty()); } #[test] @@ -2522,9 +3060,7 @@ mod tests { .allocate_process(CallerCredential::Unauthenticated, None) .unwrap(); parent.complete_start().unwrap(); - let CreatedProcess { identity, handle } = - parent.allocate_child_process(readiness_sink()).unwrap(); - let child_id = identity.process_id; + let child_id = parent.allocate_child_process().unwrap().process_id; let child = broker .processes .read() @@ -2539,9 +3075,10 @@ mod tests { parent.duplicate_object_references_to_child(ProcessId(child_id.0 + 1), &[first], &sink), Err(BrokerError::UnknownObject) ); - // A process reference is not duplicable, so the child receives nothing. + // A signals reference is not duplicable, so the child receives nothing. + let signals = crate::signal::open(&parent, readiness_sink()).unwrap(); assert_eq!( - parent.duplicate_object_references_to_child(child_id, &[first, handle], &sink), + parent.duplicate_object_references_to_child(child_id, &[first, signals], &sink), Err(BrokerError::UnsupportedOperation) ); assert_eq!(child.references.lock().handles, []); @@ -2567,8 +3104,7 @@ mod tests { ); // A child that fails to start also releases them while its parent still holds it. - let CreatedProcess { identity, .. } = - parent.allocate_child_process(readiness_sink()).unwrap(); + let identity = parent.allocate_child_process().unwrap(); parent .duplicate_object_references_to_child(identity.process_id, &[first], &sink) .unwrap(); diff --git a/litebox_broker_core/src/process_group.rs b/litebox_broker_core/src/process_group.rs new file mode 100644 index 000000000..0d45b284b --- /dev/null +++ b/litebox_broker_core/src/process_group.rs @@ -0,0 +1,286 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Process groups and sessions. +//! +//! Every process belongs to one process group, and every process group to one +//! session, each identified by the ID of the process that created it. A child +//! process starts in its creator's group and session, while a root process +//! leads its own. A group or session exists while any process, including a +//! zombie not yet reaped, belongs to it, and its ID is not reused until then. + +use alloc::sync::Arc; + +use litebox_broker_protocol::ProcessId; +use litebox_broker_protocol::process_group::ProcessGroupMembership; + +use crate::{BrokerError, BrokerProcess, Result}; + +/// Moves the process `target` into `process_group`, creating the group if it +/// is `target`'s ID. +/// +/// `target` must be the caller or one of its children, or this returns +/// `UnknownObject`. Returns `PolicyDenied` if `target` is in another session +/// than the caller or leads a session, or if `process_group` is neither +/// `target`'s ID nor an existing group in the caller's session. +pub fn set(process: &BrokerProcess, target: ProcessId, process_group: ProcessId) -> Result<()> { + let _tree = process.core.process_tree.lock(); + let target = process.core.registered_process(target)?; + if !core::ptr::eq(Arc::as_ptr(&target), process) && !target.is_child_of(process) { + return Err(BrokerError::UnknownObject); + } + let session = process.membership().session; + let membership = target.membership(); + if membership.session != session || membership.session == target.id() { + return Err(BrokerError::PolicyDenied); + } + let joined = ProcessGroupMembership { + process_group, + session, + }; + if process_group != target.id() + && !process + .core + .registered_processes() + .iter() + .any(|member| member.membership() == joined) + { + return Err(BrokerError::PolicyDenied); + } + target.set_membership(joined); + Ok(()) +} + +/// Makes the process `target` the leader of a new session and of a new process +/// group in it. +/// +/// `target` must be the caller or its pending child, or this returns +/// `UnknownObject`. Returns `PolicyDenied` if a process group already has +/// `target`'s ID. +pub fn create_session(process: &BrokerProcess, target: ProcessId) -> Result<()> { + let _tree = process.core.process_tree.lock(); + let processes = process.core.registered_processes(); + let create = |target: &BrokerProcess| { + let id = target.id(); + if processes + .iter() + .any(|member| member.membership().process_group == id) + { + return Err(BrokerError::PolicyDenied); + } + target.set_membership(ProcessGroupMembership { + process_group: id, + session: id, + }); + Ok(()) + }; + if target == process.id() { + create(process) + } else { + // The child cannot start, and so gain children in its old session, + // before it moves. + process.with_pending_child(target, |child| create(child))? + } +} + +#[cfg(test)] +mod tests { + use alloc::sync::Arc; + + use litebox_broker_protocol::ProcessId; + use litebox_broker_protocol::process::ProcessExitStatus; + use litebox_broker_protocol::process_group::ProcessGroupMembership; + + use crate::id::IdAllocator; + use crate::test_support::TestBrokerCoreBuilder; + use crate::{ + BrokerCore, BrokerError, BrokerProcess, CallerCredential, ObjectRights, PolicyEngine, + Result, + }; + + fn broker() -> BrokerCore { + TestBrokerCoreBuilder::new( + PolicyEngine::with_unauthenticated_rights(ObjectRights::all()) + .with_process_duplication_enabled(true), + ) + .build() + .unwrap() + } + + fn process(broker: &BrokerCore, parent: Option<&BrokerProcess>) -> Arc { + broker + .create_process( + CallerCredential::Unauthenticated, + parent.map(BrokerProcess::id), + ) + .unwrap() + } + + fn get(process: &BrokerProcess, target: ProcessId) -> Result { + process.process_info(target).map(|info| info.membership) + } + + fn membership( + process_group: &BrokerProcess, + session: &BrokerProcess, + ) -> ProcessGroupMembership { + ProcessGroupMembership { + process_group: process_group.id(), + session: session.id(), + } + } + + #[test] + fn children_inherit_their_parents_membership() { + let broker = broker(); + let root = process(&broker, None); + let child = process(&broker, Some(&root)); + let other = process(&broker, None); + + assert_eq!(get(&other, root.id()), Ok(membership(&root, &root))); + assert_eq!(get(&other, child.id()), Ok(membership(&root, &root))); + assert_eq!(get(&root, other.id()), Ok(membership(&other, &other))); + assert_eq!( + get(&root, ProcessId(u32::MAX)), + Err(BrokerError::UnknownObject) + ); + + super::set(&root, child.id(), child.id()).unwrap(); + let grandchild = process(&broker, Some(&child)); + assert_eq!(get(&root, grandchild.id()), Ok(membership(&child, &root))); + } + + #[test] + fn setting_groups_follows_the_session_rules() { + let broker = broker(); + let root = process(&broker, None); + let first = process(&broker, Some(&root)); + let second = process(&broker, Some(&root)); + let grandchild = process(&broker, Some(&first)); + let other = process(&broker, None); + + // Only the caller and its children can move, and not session leaders. + assert_eq!( + super::set(&root, root.id(), root.id()), + Err(BrokerError::PolicyDenied) + ); + assert_eq!( + super::set(&root, grandchild.id(), grandchild.id()), + Err(BrokerError::UnknownObject) + ); + assert_eq!( + super::set(&root, other.id(), other.id()), + Err(BrokerError::UnknownObject) + ); + + // A process creates its own group or joins one in its session. + super::set(&root, first.id(), first.id()).unwrap(); + super::set(&root, second.id(), first.id()).unwrap(); + assert_eq!(get(&root, second.id()), Ok(membership(&first, &root))); + super::set(&second, second.id(), root.id()).unwrap(); + assert_eq!(get(&root, second.id()), Ok(membership(&root, &root))); + assert_eq!( + super::set(&second, second.id(), ProcessId(u32::MAX)), + Err(BrokerError::PolicyDenied) + ); + assert_eq!( + super::set(&second, second.id(), other.id()), + Err(BrokerError::PolicyDenied) + ); + super::set(&first, grandchild.id(), root.id()).unwrap(); + + // A child in another session cannot move, nor can a session leader. + super::create_session(&grandchild, grandchild.id()).unwrap(); + assert_eq!( + super::set(&first, grandchild.id(), first.id()), + Err(BrokerError::PolicyDenied) + ); + assert_eq!( + super::set(&grandchild, grandchild.id(), grandchild.id()), + Err(BrokerError::PolicyDenied) + ); + let great_grandchild = process(&broker, Some(&grandchild)); + assert_eq!( + super::set(&great_grandchild, great_grandchild.id(), root.id()), + Err(BrokerError::PolicyDenied) + ); + super::set(&great_grandchild, great_grandchild.id(), grandchild.id()).unwrap(); + } + + #[test] + fn sessions_are_created_by_processes_not_leading_a_group() { + let broker = broker(); + let root = process(&broker, None); + root.complete_start().unwrap(); + let child = process(&broker, Some(&root)); + + assert_eq!( + super::create_session(&root, root.id()), + Err(BrokerError::PolicyDenied) + ); + // Only the caller and its pending child can be targeted. + assert_eq!( + super::create_session(&root, child.id()), + Err(BrokerError::UnknownObject) + ); + let pending = root.allocate_child_process().unwrap().process_id; + super::create_session(&root, pending).unwrap(); + assert_eq!( + get(&root, pending), + Ok(ProcessGroupMembership { + process_group: pending, + session: pending, + }) + ); + + // A group keeps its ID while another process is in it. + super::set(&root, child.id(), child.id()).unwrap(); + let grandchild = process(&broker, Some(&child)); + super::set(&child, child.id(), root.id()).unwrap(); + assert_eq!( + super::create_session(&child, child.id()), + Err(BrokerError::PolicyDenied) + ); + super::set(&child, grandchild.id(), root.id()).unwrap(); + super::create_session(&child, child.id()).unwrap(); + assert_eq!(get(&root, child.id()), Ok(membership(&child, &child))); + } + + #[test] + fn group_and_session_ids_are_not_reused_while_in_use() { + let mut broker = broker(); + broker.ids = Arc::new(spin::Mutex::new(IdAllocator::new(3).unwrap())); + let allocate = |parent: Option<&BrokerProcess>| { + broker.allocate_process( + CallerCredential::Unauthenticated, + parent.map(BrokerProcess::id), + ) + }; + let leader = allocate(None).unwrap(); + leader.complete_start().unwrap(); + let member = allocate(Some(&leader)).unwrap(); + member.complete_start().unwrap(); + let leader_id = leader.id(); + leader.retire(true); + leader + .complete_exit(ProcessExitStatus::Exited { code: 0 }) + .unwrap(); + drop(leader); + let other = allocate(None).unwrap(); + + // The leader's ID stays allocated while its group and session exist. + assert_eq!( + allocate(None).map(|process| process.id()), + Err(BrokerError::ResourceExhausted) + ); + assert_eq!( + get(&other, member.id()), + Ok(ProcessGroupMembership { + process_group: leader_id, + session: leader_id, + }) + ); + super::create_session(&member, member.id()).unwrap(); + assert_eq!(allocate(None).unwrap().id(), leader_id); + } +} diff --git a/litebox_broker_core/src/signal.rs b/litebox_broker_core/src/signal.rs index 41225294c..12a76d891 100644 --- a/litebox_broker_core/src/signal.rs +++ b/litebox_broker_core/src/signal.rs @@ -1,17 +1,20 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT license. -//! Signals sent between broker processes. +//! Signals sent between broker processes, and child event notifications. //! -//! Every process has pending signals that any process may send to. A process -//! opens a handle to take them, which becomes readable while any is pending. -//! Signals sent before the process opens its handle stay pending until it -//! takes them. +//! Every process has pending signals that any process may send to, and learns +//! of its children's exits and removals alongside them. A process opens a handle to take +//! these events, which becomes readable while any is pending. Events that +//! occur before the process opens its handle stay pending until it takes +//! them. -use alloc::sync::{Arc, Weak}; +use alloc::sync::Arc; +use alloc::vec; +use litebox_broker_protocol::process::ChildExit; use litebox_broker_protocol::readiness::ReadinessFlags; -use litebox_broker_protocol::signal::{MAX_SIGNAL, PendingSignal}; +use litebox_broker_protocol::signal::{MAX_SIGNAL, PendingSignal, SignalEvent, SignalTarget}; use litebox_broker_protocol::{ObjectHandle, ProcessId}; use spin::Mutex; @@ -20,7 +23,7 @@ use crate::readiness::{ReadinessRegistration, ReadinessSink}; use crate::{BrokerError, BrokerProcess, Result}; /// Opens the caller's signals, returning a handle that becomes readable while -/// any is pending. +/// any event is pending. /// /// A process may hold only one such handle at a time; opening another fails /// with `ResourceExhausted`. Fails with `PolicyDenied` if the caller may not @@ -50,34 +53,46 @@ pub fn open( })) } -/// Sends `signal` to the process `target`, or only checks that it exists if -/// `signal` is zero. +/// Sends `signal` to the processes `target` selects, or only checks that one +/// exists if `signal` is zero. /// -/// A signal already pending for the target is not sent again. Returns -/// `UnknownObject` if no such process exists. -pub fn send(process: &BrokerProcess, target: ProcessId, signal: u32) -> Result<()> { +/// A signal already pending for a target is not sent again. Returns +/// `UnknownObject` if no process is targeted. +pub fn send(process: &BrokerProcess, target: SignalTarget, signal: u32) -> Result<()> { if signal > MAX_SIGNAL { return Err(BrokerError::UnsupportedOperation); } - // The registry lock is released before the target can drop, since a final - // process drop removes itself from the registry. - let target = process - .core - .processes - .read() - .get(&target) - .and_then(Weak::upgrade) - .ok_or(BrokerError::UnknownObject)?; + let targets = match target { + SignalTarget::Process(id) => vec![process.core.registered_process(id)?], + SignalTarget::ProcessGroup(process_group) => { + // Members cannot move between groups while they are selected. + let _tree = process.core.process_tree.lock(); + let mut targets = process.core.registered_processes(); + targets.retain(|target| target.membership().process_group == process_group); + targets + } + SignalTarget::All => { + let mut targets = process.core.registered_processes(); + targets.retain(|target| target.id() != process.id() && target.has_creator()); + targets + } + }; + if targets.is_empty() { + return Err(BrokerError::UnknownObject); + } if let Some(index) = signal.checked_sub(1) { - target.signals.send(index as usize, process.id); + for target in &targets { + target.signals.send(index as usize, process.id); + } } Ok(()) } -/// Takes the caller's lowest-numbered pending signal. +/// Takes the caller's lowest-numbered pending signal, or else its pending +/// child exit, or else its pending child removal. /// -/// Returns `WouldBlock` if none is pending. -pub fn take(process: &BrokerProcess, handle: ObjectHandle) -> Result { +/// Returns `WouldBlock` if no event is pending. +pub fn take(process: &BrokerProcess, handle: ObjectHandle) -> Result { let object = process.authorized_object(handle, ObjectRights::WAIT)?; let object = object.read(); object.as_signals()?.signals.take() @@ -92,13 +107,21 @@ impl ObjectEntry { } } -/// The pending signals of one process. +/// The pending signals and child events of one process. pub(crate) struct ProcessSignals(Mutex); struct ProcessSignalsState { /// The process that first sent each pending signal, indexed by signal /// number minus one. senders: [Option; MAX_SIGNAL as usize], + /// The earliest child exit since the process last took one. + /// + /// Like a pending signal, later exits coalesce into it. The process + /// reaps its children separately, so only the notification coalesces. + child_exit: Option, + /// Whether a child left the process without exiting since the process + /// last took this event. + child_removed: bool, /// Publication for the process's open handle, if any. readiness: Option, } @@ -107,6 +130,8 @@ impl ProcessSignals { pub(crate) fn new() -> Self { Self(Mutex::new(ProcessSignalsState { senders: [None; MAX_SIGNAL as usize], + child_exit: None, + child_removed: false, readiness: None, })) } @@ -117,29 +142,59 @@ impl ProcessSignals { return; } state.senders[index] = Some(sender); - // Republish, since a waiter that took the previous signals may not - // have observed readiness change since. - if let Some(readiness) = &state.readiness { - let _ = readiness.republish(ReadinessFlags::READ); + state.republish(); + } + + /// Notifies the process that its child exited, unless an earlier exit + /// is still pending. + pub(crate) fn post_child_exit(&self, exit: ChildExit) { + let mut state = self.0.lock(); + if state.child_exit.is_some() { + return; + } + state.child_exit = Some(exit); + state.republish(); + } + + /// Notifies the process that a child left it without exiting, unless an + /// earlier removal is still pending. + pub(crate) fn post_child_removed(&self) { + let mut state = self.0.lock(); + if state.child_removed { + return; } + state.child_removed = true; + state.republish(); } - fn take(&self) -> Result { + fn take(&self) -> Result { let mut state = self.0.lock(); - let (index, sender) = state + let signal = state .senders .iter_mut() .enumerate() - .find_map(|(index, sender)| Some((index, sender.take()?))) - .ok_or(BrokerError::WouldBlock)?; - Ok(PendingSignal { - signal: u32::try_from(index + 1).expect("signal numbers fit in u32"), - sender, - }) + .find_map(|(index, sender)| Some((index, sender.take()?))); + if let Some((index, sender)) = signal { + return Ok(SignalEvent::Signal(PendingSignal { + signal: u32::try_from(index + 1).expect("signal numbers fit in u32"), + sender, + })); + } + if let Some(exit) = state.child_exit.take() { + return Ok(SignalEvent::ChildExited(exit)); + } + if core::mem::take(&mut state.child_removed) { + return Ok(SignalEvent::ChildRemoved); + } + Err(BrokerError::WouldBlock) } fn readiness(&self) -> ReadinessFlags { - if self.0.lock().senders.iter().any(Option::is_some) { + let state = self.0.lock(); + if state.child_exit.is_some() + || state.child_removed + || state.senders.iter().any(Option::is_some) + { ReadinessFlags::READ } else { ReadinessFlags::default() @@ -147,6 +202,17 @@ impl ProcessSignals { } } +impl ProcessSignalsState { + /// Republishes readiness after an event becomes pending, since a waiter + /// that took the previous events may not have observed readiness change + /// since. + fn republish(&self) { + if let Some(readiness) = &self.readiness { + let _ = readiness.republish(ReadinessFlags::READ); + } + } +} + /// A process's handle to its own pending signals. pub(crate) struct SignalsObject { signals: Arc, @@ -171,8 +237,9 @@ impl Drop for SignalsObject { mod tests { use alloc::sync::Arc; + use litebox_broker_protocol::process::{ChildExit, ProcessExitStatus}; use litebox_broker_protocol::readiness::ReadinessFlags; - use litebox_broker_protocol::signal::PendingSignal; + use litebox_broker_protocol::signal::{PendingSignal, SignalEvent, SignalTarget}; use litebox_broker_protocol::{ObjectHandle, ProcessId}; use crate::readiness::tests::TestReadinessSink; @@ -195,8 +262,12 @@ mod tests { .unwrap() } - fn pending(signal: u32, sender: ProcessId) -> PendingSignal { - PendingSignal { signal, sender } + fn pending(signal: u32, sender: ProcessId) -> SignalEvent { + SignalEvent::Signal(PendingSignal { signal, sender }) + } + + fn to(process: &BrokerProcess) -> SignalTarget { + SignalTarget::Process(process.id()) } #[test] @@ -208,10 +279,10 @@ mod tests { // Signals sent before the target opens its handle stay pending, and // repeated signals keep their first sender. - super::send(&first, target.id(), 10).unwrap(); - super::send(&second, target.id(), 10).unwrap(); - super::send(&second, target.id(), 64).unwrap(); - super::send(&second, target.id(), 2).unwrap(); + super::send(&first, to(&target), 10).unwrap(); + super::send(&second, to(&target), 10).unwrap(); + super::send(&second, to(&target), 64).unwrap(); + super::send(&second, to(&target), 2).unwrap(); let sink = Arc::new(TestReadinessSink::default()); let handle = super::open(&target, sink.clone()).unwrap(); assert_eq!(target.check_readiness(handle), Ok(ReadinessFlags::READ)); @@ -227,8 +298,8 @@ mod tests { // A signal sent while the handle is open republishes readiness, but // one already pending does not. - super::send(&first, target.id(), 15).unwrap(); - super::send(&second, target.id(), 15).unwrap(); + super::send(&first, to(&target), 15).unwrap(); + super::send(&second, to(&target), 15).unwrap(); assert_eq!( *sink.republished.lock().unwrap(), [(handle, ReadinessFlags::READ)] @@ -241,7 +312,7 @@ mod tests { target.close_object_reference(handle).unwrap(); assert_eq!(*sink.retired.lock().unwrap(), [handle]); - super::send(&first, target.id(), 15).unwrap(); + super::send(&first, to(&target), 15).unwrap(); assert_eq!(sink.republished.lock().unwrap().len(), 1); } @@ -253,27 +324,153 @@ mod tests { let handle = super::open(&target, Arc::new(TestReadinessSink::default())).unwrap(); // Signal zero only checks that the target exists. - super::send(&sender, target.id(), 0).unwrap(); + super::send(&sender, to(&target), 0).unwrap(); assert_eq!(super::take(&target, handle), Err(BrokerError::WouldBlock)); assert_eq!( - super::send(&sender, ProcessId(u32::MAX), 0), + super::send(&sender, SignalTarget::Process(ProcessId(u32::MAX)), 0), Err(BrokerError::UnknownObject) ); assert_eq!( - super::send(&sender, ProcessId(u32::MAX), 9), + super::send(&sender, SignalTarget::Process(ProcessId(u32::MAX)), 9), Err(BrokerError::UnknownObject) ); assert_eq!( - super::send(&sender, target.id(), 65), + super::send(&sender, to(&target), 65), Err(BrokerError::UnsupportedOperation) ); assert_eq!(super::take(&target, handle), Err(BrokerError::WouldBlock)); // A process can signal itself. - super::send(&target, target.id(), 1).unwrap(); + super::send(&target, to(&target), 1).unwrap(); assert_eq!(super::take(&target, handle), Ok(pending(1, target.id()))); } + #[test] + fn signals_reach_process_groups_and_all_processes() { + let broker = broker(); + let root = process(&broker); + let other = process(&broker); + let first = broker + .create_process(CallerCredential::Unauthenticated, Some(root.id())) + .unwrap(); + // No process other than the caller has a creator yet. + assert_eq!( + super::send(&first, SignalTarget::All, 0), + Err(BrokerError::UnknownObject) + ); + let second = broker + .create_process(CallerCredential::Unauthenticated, Some(root.id())) + .unwrap(); + crate::process_group::set(&root, second.id(), second.id()).unwrap(); + let sink = Arc::new(TestReadinessSink::default()); + let handles = [&root, &other, &first, &second] + .map(|process| (process, super::open(process, sink.clone()).unwrap())); + let take_all = || { + handles + .iter() + .filter_map(|(process, handle)| match super::take(process, *handle) { + Ok(SignalEvent::Signal(signal)) => Some((process.id(), signal.signal)), + _ => None, + }) + .collect::>() + }; + + super::send(&other, SignalTarget::ProcessGroup(root.id()), 10).unwrap(); + assert_eq!(take_all(), [(root.id(), 10), (first.id(), 10)]); + super::send(&other, SignalTarget::ProcessGroup(second.id()), 0).unwrap(); + super::send(&other, SignalTarget::ProcessGroup(second.id()), 12).unwrap(); + assert_eq!(take_all(), [(second.id(), 12)]); + assert_eq!( + super::send(&other, SignalTarget::ProcessGroup(ProcessId(u32::MAX)), 0), + Err(BrokerError::UnknownObject) + ); + + // Root processes and the sender are spared. + super::send(&first, SignalTarget::All, 15).unwrap(); + assert_eq!(take_all(), [(second.id(), 15)]); + super::send(&root, SignalTarget::All, 15).unwrap(); + assert_eq!(take_all(), [(first.id(), 15), (second.id(), 15)]); + assert_eq!( + super::send(&root, SignalTarget::All, 65), + Err(BrokerError::UnsupportedOperation) + ); + } + + #[test] + fn child_exits_coalesce_and_follow_pending_signals() { + let broker = broker(); + let parent = process(&broker); + let sink = Arc::new(TestReadinessSink::default()); + let handle = super::open(&parent, sink.clone()).unwrap(); + let exit = |id, code| ChildExit { + process_id: ProcessId(id), + exit_status: ProcessExitStatus::Exited { code }, + }; + + parent.signals.post_child_exit(exit(100, 1)); + parent.signals.post_child_exit(exit(101, 2)); + super::send(&parent, to(&parent), 17).unwrap(); + assert_eq!( + *sink.republished.lock().unwrap(), + [ + (handle, ReadinessFlags::READ), + (handle, ReadinessFlags::READ) + ] + ); + assert_eq!(super::take(&parent, handle), Ok(pending(17, parent.id()))); + assert_eq!(parent.check_readiness(handle), Ok(ReadinessFlags::READ)); + assert_eq!( + super::take(&parent, handle), + Ok(SignalEvent::ChildExited(exit(100, 1))) + ); + assert_eq!(super::take(&parent, handle), Err(BrokerError::WouldBlock)); + assert_eq!( + parent.check_readiness(handle), + Ok(ReadinessFlags::default()) + ); + + // Once taken, the next exit is notified again. + parent.signals.post_child_exit(exit(102, 3)); + assert_eq!(sink.republished.lock().unwrap().len(), 3); + assert_eq!( + super::take(&parent, handle), + Ok(SignalEvent::ChildExited(exit(102, 3))) + ); + } + + #[test] + fn child_removals_coalesce_and_follow_child_exits() { + let broker = broker(); + let parent = process(&broker); + let sink = Arc::new(TestReadinessSink::default()); + let handle = super::open(&parent, sink.clone()).unwrap(); + let exit = ChildExit { + process_id: ProcessId(100), + exit_status: ProcessExitStatus::Exited { code: 1 }, + }; + + parent.signals.post_child_removed(); + parent.signals.post_child_removed(); + parent.signals.post_child_exit(exit); + assert_eq!(sink.republished.lock().unwrap().len(), 2); + assert_eq!( + super::take(&parent, handle), + Ok(SignalEvent::ChildExited(exit)) + ); + assert_eq!(parent.check_readiness(handle), Ok(ReadinessFlags::READ)); + assert_eq!(super::take(&parent, handle), Ok(SignalEvent::ChildRemoved)); + assert_eq!(super::take(&parent, handle), Err(BrokerError::WouldBlock)); + assert_eq!( + parent.check_readiness(handle), + Ok(ReadinessFlags::default()) + ); + + // Once taken, the next removal is notified again. + parent.signals.post_child_removed(); + assert_eq!(sink.republished.lock().unwrap().len(), 3); + assert_eq!(super::take(&parent, handle), Ok(SignalEvent::ChildRemoved)); + } + #[test] fn a_process_opens_one_signals_handle_at_a_time() { let broker = broker(); @@ -289,7 +486,7 @@ mod tests { let retired = sink.retired.lock().unwrap().clone(); assert_eq!(retired.len(), 1); assert_ne!(retired[0], handle); - super::send(&process, process.id(), 3).unwrap(); + super::send(&process, to(&process), 3).unwrap(); assert_eq!( *sink.republished.lock().unwrap(), [(handle, ReadinessFlags::READ)] diff --git a/litebox_broker_core/src/socket/tests.rs b/litebox_broker_core/src/socket/tests.rs index bd9b1f2bd..c3cfc000e 100644 --- a/litebox_broker_core/src/socket/tests.rs +++ b/litebox_broker_core/src/socket/tests.rs @@ -1329,6 +1329,7 @@ fn test_broker_with_policy( crate::id::IdAllocator::new(crate::id::MAX_ALLOCATED_ID).unwrap(), )), processes: Arc::new(spin::RwLock::new(hashbrown::HashMap::new())), + process_tree: Arc::new(spin::Mutex::new(())), active_thread_count: Arc::new(AtomicUsize::new(0)), next_reference_handle: Arc::new(spin::RwLock::new(1)), references: Arc::new(spin::RwLock::new(hashbrown::HashMap::new())), diff --git a/litebox_broker_host/src/lib.rs b/litebox_broker_host/src/lib.rs index 6d604696d..763541f17 100644 --- a/litebox_broker_host/src/lib.rs +++ b/litebox_broker_host/src/lib.rs @@ -40,7 +40,8 @@ use litebox_broker_protocol::fs::{ use litebox_broker_protocol::message::{ BrokerHandshakeResponse, BrokerOperation, BrokerRequest, BrokerResponse, BrokerResult, EventRequest, EventResponse, FileRequest, FileResponse, PipeRequest, PipeResponse, - SignalRequest, SignalResponse, SocketRequest, SocketResponse, TimerRequest, TimerResponse, + ProcessGroupRequest, ProcessGroupResponse, SignalRequest, SignalResponse, SocketRequest, + SocketResponse, TimerRequest, TimerResponse, }; use litebox_broker_protocol::pipe::{ CreatePipeResponse, MAX_PIPE_TRANSFER_SIZE, ReadPipeResponse, WritePipeResponse, @@ -265,14 +266,13 @@ where None => None, }; let limits = core.limits(); - // Sockets, child process handles, and object references (shared pipes, - // timers, and files such as stdio devices) register readiness. Add future - // resource limits here so every live registration fits in the - // association's shared readiness sink. + // Sockets and object references (shared pipes, timers, signals, and files + // such as stdio devices) register readiness. Add future resource limits + // here so every live registration fits in the association's shared + // readiness sink. let max_live_readiness_registrations = limits .max_sockets .min(limits.max_sockets_per_process) - .saturating_add(limits.max_processes) .saturating_add(limits.max_references.min(limits.max_references_per_process)); if max_live_readiness_registrations > readiness_sink.max_tracked_objects() { return Err(BrokerHostError::Broker(ErrorCode::ResourceExhausted)); @@ -500,7 +500,7 @@ fn handle_request( .map(BrokerResult::CreateThread) .map_err(RequestFailure::from), CreateThreadRequest::Process => process - .allocate_child_process(Arc::clone(readiness_sink)) + .allocate_child_process() .map(CreateThreadResponse::Process) .map(BrokerResult::CreateThread) .map_err(RequestFailure::from), @@ -525,9 +525,9 @@ fn handle_request( .set_status_flags(request) .map(|()| BrokerResult::StatusFlagsSet) .map_err(RequestFailure::from), - BrokerOperation::GetProcessExitStatus(handle) => process - .process_exit_status(handle) - .map(BrokerResult::ProcessExitStatus) + BrokerOperation::ReapChild(selector) => process + .reap_child(selector) + .map(BrokerResult::ChildReaped) .map_err(RequestFailure::from), BrokerOperation::ExitChildProcess(ExitChildProcessRequest { child_process_id, @@ -536,6 +536,10 @@ fn handle_request( .exit_child_process(child_process_id, exit_status) .map(|()| BrokerResult::ProcessExited) .map_err(RequestFailure::from), + BrokerOperation::CancelChildProcess(child_process_id) => process + .cancel_child_process(child_process_id) + .map(|()| BrokerResult::ChildProcessCancelled) + .map_err(RequestFailure::from), BrokerOperation::ReportExitStatus(exit_status) => process .report_exit_status(exit_status) .map(|()| BrokerResult::ExitStatusReported) @@ -544,6 +548,14 @@ fn handle_request( .set_child_reaping(enabled) .map(|()| BrokerResult::ChildReapingSet) .map_err(RequestFailure::from), + BrokerOperation::SetOrphanAdoption(enabled) => process + .set_orphan_adoption(enabled) + .map(|()| BrokerResult::OrphanAdoptionSet) + .map_err(RequestFailure::from), + BrokerOperation::GetProcessInfo(process_id) => process + .process_info(process_id) + .map(BrokerResult::ProcessInfo) + .map_err(RequestFailure::from), BrokerOperation::DuplicateObjectsToChild(request) => { duplicate_objects_to_child(process, request, shared_buffers, readiness_sink) .map(|()| BrokerResult::ObjectsDuplicated) @@ -560,6 +572,9 @@ fn handle_request( BrokerOperation::Signal(request) => { handle_signal_request(process, request, readiness_sink).map(BrokerResult::Signal) } + BrokerOperation::ProcessGroup(request) => { + handle_process_group_request(process, request).map(BrokerResult::ProcessGroup) + } BrokerOperation::Socket(request) => { handle_socket_request(process, request, shared_buffers, readiness_sink) .map(BrokerResult::Socket) @@ -1398,7 +1413,7 @@ fn handle_signal_request( .map(|handle| SignalResponse::Open(OpenSignalsResponse { handle })) } SignalRequest::Send(request) => { - litebox_broker_core::signal::send(process, request.process_id, request.signal) + litebox_broker_core::signal::send(process, request.target, request.signal) .map(|()| SignalResponse::Sent) } SignalRequest::Take(request) => { @@ -1408,6 +1423,25 @@ fn handle_signal_request( response.map_err(RequestFailure::from) } +fn handle_process_group_request( + process: &BrokerProcess, + request: ProcessGroupRequest, +) -> RequestResult { + let response = match request { + ProcessGroupRequest::Set(request) => litebox_broker_core::process_group::set( + process, + request.process_id, + request.process_group, + ) + .map(|()| ProcessGroupResponse::Set), + ProcessGroupRequest::CreateSession(process_id) => { + litebox_broker_core::process_group::create_session(process, process_id) + .map(|()| ProcessGroupResponse::CreateSession) + } + }; + response.map_err(RequestFailure::from) +} + /// Terminal outcome after processing one broker connection. #[derive(Clone, Copy, Debug, PartialEq, Eq)] #[non_exhaustive] @@ -1447,11 +1481,16 @@ mod tests { }; use litebox_broker_protocol::message::BrokerHandshakeRequest; use litebox_broker_protocol::pipe::{CreatePipeRequest, ReadPipeRequest, WritePipeRequest}; + use litebox_broker_protocol::process::{ + ChildExit, ChildSelector, ProcessExitStatus, ProcessInfo, + }; + use litebox_broker_protocol::process_group::{ProcessGroupMembership, SetProcessGroupRequest}; use litebox_broker_protocol::random::MAX_RANDOM_TRANSFER_SIZE; use litebox_broker_protocol::shared_buffer::{ SHARED_BUFFER_LAYOUT, SHARED_BUFFER_POOL_SIZE, SHARED_BUFFER_SLOT_SIZE, SharedBufferSlotIndex, }; + use litebox_broker_protocol::signal::{SendSignalRequest, SignalTarget}; use litebox_broker_protocol::socket::{ AddressFamily, ConnectSocketRequest, CreateSocketRequest, IpProtocol, ReceiveFlags, ReceiveFromFlags, ReceiveFromSocketRequest, ReceiveFromSocketResponse, @@ -1800,6 +1839,8 @@ mod tests { active_request_allocates_and_releases_thread_id(&broker); active_request_closes_object_reference(&broker); active_requests_operate_timers(&broker, &timer_provider); + active_requests_manage_process_groups(&broker); + active_requests_manage_children(&broker); association_shared_buffer_sequences_stage_pipe_data(&broker); association_shared_buffer_sequences_stage_socket_data(&broker); association_shared_buffer_sequence_stages_random_data(&broker); @@ -2576,6 +2617,110 @@ mod tests { assert_eq!(clock.alarm_count(), 0); } + fn active_requests_manage_process_groups(broker: &BrokerCore) { + let parent = broker + .create_process(CallerCredential::Unauthenticated, None) + .unwrap(); + parent.complete_start().unwrap(); + let child = broker + .create_process(CallerCredential::Unauthenticated, Some(parent.id())) + .unwrap(); + let request = |process: &BrokerProcess, request| { + handle_test_request(process, BrokerOperation::ProcessGroup(request)) + }; + let info = |process_group: &BrokerProcess| { + BrokerResult::ProcessInfo(ProcessInfo { + creator: Some(parent.id()), + parent: Some(parent.id()), + membership: ProcessGroupMembership { + process_group: process_group.id(), + session: parent.id(), + }, + }) + }; + + assert_eq!( + handle_test_request(&child, BrokerOperation::GetProcessInfo(child.id())), + info(&parent) + ); + assert_eq!( + request( + &parent, + ProcessGroupRequest::Set(SetProcessGroupRequest { + process_id: child.id(), + process_group: child.id(), + }) + ), + BrokerResult::ProcessGroup(ProcessGroupResponse::Set) + ); + assert_eq!( + handle_test_request(&parent, BrokerOperation::GetProcessInfo(child.id())), + info(&child) + ); + assert_eq!( + request(&child, ProcessGroupRequest::CreateSession(child.id())), + BrokerResult::Error(ErrorCode::PolicyDenied) + ); + assert_eq!( + request(&parent, ProcessGroupRequest::CreateSession(child.id())), + BrokerResult::Error(ErrorCode::UnknownObject) + ); + assert_eq!( + handle_test_request( + &parent, + BrokerOperation::Signal(SignalRequest::Send(SendSignalRequest { + target: SignalTarget::ProcessGroup(child.id()), + signal: 0, + })) + ), + BrokerResult::Signal(SignalResponse::Sent) + ); + } + + fn active_requests_manage_children(broker: &BrokerCore) { + let parent = broker + .create_process(CallerCredential::Unauthenticated, None) + .unwrap(); + parent.complete_start().unwrap(); + let child = broker + .create_process(CallerCredential::Unauthenticated, Some(parent.id())) + .unwrap(); + child.complete_start().unwrap(); + let reap = |selector| handle_test_request(&parent, BrokerOperation::ReapChild(selector)); + + assert_eq!( + reap(ChildSelector::Any), + BrokerResult::Error(ErrorCode::WouldBlock) + ); + assert_eq!( + handle_test_request(&parent, BrokerOperation::SetOrphanAdoption(true)), + BrokerResult::OrphanAdoptionSet + ); + assert_eq!( + handle_test_request(&parent, BrokerOperation::CancelChildProcess(child.id())), + BrokerResult::Error(ErrorCode::UnknownObject) + ); + child.retire(true); + child + .complete_exit(ProcessExitStatus::Exited { code: 3 }) + .unwrap(); + assert_eq!( + reap(ChildSelector::Process(child.id())), + BrokerResult::ChildReaped(ChildExit { + process_id: child.id(), + exit_status: ProcessExitStatus::Exited { code: 3 }, + }) + ); + assert_eq!( + reap(ChildSelector::Any), + BrokerResult::Error(ErrorCode::UnknownObject) + ); + assert_eq!( + handle_test_request(&parent, BrokerOperation::GetProcessInfo(child.id())), + BrokerResult::Error(ErrorCode::UnknownObject) + ); + } + fn active_request_allocates_and_releases_thread_id(broker: &BrokerCore) { let process = broker .create_process(CallerCredential::Unauthenticated, None) diff --git a/litebox_broker_local/src/lib.rs b/litebox_broker_local/src/lib.rs index e15d7396b..2dc4ad6e4 100644 --- a/litebox_broker_local/src/lib.rs +++ b/litebox_broker_local/src/lib.rs @@ -24,6 +24,7 @@ mod error; mod event; mod fs; mod pipe; +mod process_group; mod random; mod signal; mod socket; @@ -42,10 +43,11 @@ use litebox_broker_protocol::message::{ BrokerRequest, BrokerResponse, BrokerResult, }; use litebox_broker_protocol::process::{ - CreateThreadRequest, CreateThreadResponse, CreatedProcess, DuplicateObjectsToChildRequest, - ExitChildProcessRequest, MAX_CHILD_MEMORY_WRITE_SIZE, MAX_CHILD_OBJECT_DUPLICATES, - MAX_PROCESS_BOOTSTRAP_SIZE, ProcessExitStatus, ProcessStartupData, ProcessStartupDescriptor, - ProcessTermination, StartChildProcessRequest, WriteChildMemoryRequest, + ChildExit, ChildSelector, CreateThreadRequest, CreateThreadResponse, + DuplicateObjectsToChildRequest, ExitChildProcessRequest, MAX_CHILD_MEMORY_WRITE_SIZE, + MAX_CHILD_OBJECT_DUPLICATES, MAX_PROCESS_BOOTSTRAP_SIZE, ProcessExitStatus, ProcessIdentity, + ProcessInfo, ProcessStartupData, ProcessStartupDescriptor, StartChildProcessRequest, + WriteChildMemoryRequest, }; use litebox_broker_protocol::readiness::ReadinessFlags; use litebox_broker_protocol::shared_buffer::{SHARED_BUFFER_LAYOUT, SharedBufferSequence}; @@ -276,6 +278,22 @@ impl BrokerLocal { } } + /// Discards a pending child created by [`Self::allocate_child_process`] + /// that never started, as if it had never been created. + /// + /// # Panics + /// + /// Panics if the broker returns a response for another operation. + pub fn cancel_child_process(&self, child_process_id: ProcessId) -> Result<(), Channel::Error> { + match self.request(BrokerOperation::CancelChildProcess(child_process_id))? { + BrokerResult::ChildProcessCancelled => Ok(()), + BrokerResult::Error(error) => Err(BrokerLocalError::Broker(error)), + response => { + panic!("broker returned unexpected cancel-child-process response: {response:?}") + } + } + } + /// Duplicates this process's object references into a pending child /// created by [`Self::allocate_child_process`], returning the child's /// handles in the same order. @@ -359,12 +377,28 @@ impl BrokerLocal { } } - /// Allocates one pending child process and its parent-owned handle. + /// Sets whether this process adopts the orphaned children of its exiting + /// descendants. /// /// # Panics /// /// Panics if the broker returns a response for another operation. - pub fn allocate_child_process(&self) -> Result { + pub fn set_orphan_adoption(&self, enabled: bool) -> Result<(), Channel::Error> { + match self.request(BrokerOperation::SetOrphanAdoption(enabled))? { + BrokerResult::OrphanAdoptionSet => Ok(()), + BrokerResult::Error(error) => Err(BrokerLocalError::Broker(error)), + response => { + panic!("broker returned unexpected orphan-adoption response: {response:?}") + } + } + } + + /// Allocates one pending child process of this process. + /// + /// # Panics + /// + /// Panics if the broker returns a response for another operation. + pub fn allocate_child_process(&self) -> Result { match self.request(BrokerOperation::CreateThread(CreateThreadRequest::Process))? { BrokerResult::CreateThread(CreateThreadResponse::Process(child)) => Ok(child), BrokerResult::Error(error) => Err(BrokerLocalError::Broker(error)), @@ -374,23 +408,35 @@ impl BrokerLocal { } } - /// Returns a process's termination status through a process handle. + /// Reaps this process's oldest exited child that `selector` matches, + /// returning its exit. /// - /// Returns `WouldBlock` while the process is live. + /// Returns `WouldBlock` if only live children match, and `UnknownObject` + /// if none does. /// /// # Panics /// /// Panics if the broker returns a response for another operation. - pub fn process_exit_status( - &self, - handle: ObjectHandle, - ) -> Result { - match self.request(BrokerOperation::GetProcessExitStatus(handle))? { - BrokerResult::ProcessExitStatus(status) => Ok(status), + pub fn reap_child(&self, selector: ChildSelector) -> Result { + match self.request(BrokerOperation::ReapChild(selector))? { + BrokerResult::ChildReaped(exit) => Ok(exit), BrokerResult::Error(error) => Err(BrokerLocalError::Broker(error)), - response => { - panic!("broker returned unexpected process-exit-status response: {response:?}") - } + response => panic!("broker returned unexpected reap-child response: {response:?}"), + } + } + + /// Returns the place in the process tree of process `process_id`. + /// + /// Returns `UnknownObject` if no such process exists. + /// + /// # Panics + /// + /// Panics if the broker returns a response for another operation. + pub fn process_info(&self, process_id: ProcessId) -> Result { + match self.request(BrokerOperation::GetProcessInfo(process_id))? { + BrokerResult::ProcessInfo(info) => Ok(info), + BrokerResult::Error(error) => Err(BrokerLocalError::Broker(error)), + response => panic!("broker returned unexpected process-info response: {response:?}"), } } diff --git a/litebox_broker_local/src/process_group.rs b/litebox_broker_local/src/process_group.rs new file mode 100644 index 000000000..beb243165 --- /dev/null +++ b/litebox_broker_local/src/process_group.rs @@ -0,0 +1,64 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use litebox_broker_protocol::ProcessId; +use litebox_broker_protocol::message::{ + BrokerOperation, BrokerResult, ProcessGroupRequest, ProcessGroupResponse, +}; +use litebox_broker_protocol::process_group::SetProcessGroupRequest; +use litebox_broker_transport::channel::LocalCallChannel; + +use crate::{BrokerLocal, BrokerLocalError, Result}; + +impl BrokerLocal { + /// Moves process `process_id` into `process_group`. + /// + /// # Panics + /// + /// Panics if the broker reports an unrecoverable error or returns a protocol + /// response that does not match the issued process group request. + pub fn set_process_group( + &self, + process_id: ProcessId, + process_group: ProcessId, + ) -> Result<(), Channel::Error> { + match self.request_process_group(ProcessGroupRequest::Set(SetProcessGroupRequest { + process_id, + process_group, + }))? { + ProcessGroupResponse::Set => Ok(()), + response @ ProcessGroupResponse::CreateSession => { + panic!("broker returned unexpected process group response: {response:?}") + } + } + } + + /// Makes process `process_id` the leader of a new session and of a new + /// process group in it. + /// + /// # Panics + /// + /// Panics if the broker reports an unrecoverable error or returns a protocol + /// response that does not match the issued process group request. + pub fn create_session(&self, process_id: ProcessId) -> Result<(), Channel::Error> { + match self.request_process_group(ProcessGroupRequest::CreateSession(process_id))? { + ProcessGroupResponse::CreateSession => Ok(()), + response @ ProcessGroupResponse::Set => { + panic!("broker returned unexpected process group response: {response:?}") + } + } + } + + fn request_process_group( + &self, + request: ProcessGroupRequest, + ) -> Result { + match self.request(BrokerOperation::ProcessGroup(request))? { + BrokerResult::ProcessGroup(response) => Ok(response), + BrokerResult::Error(error) => Err(BrokerLocalError::Broker(error)), + response => { + panic!("broker returned unexpected process group response: {response:?}"); + } + } + } +} diff --git a/litebox_broker_local/src/signal.rs b/litebox_broker_local/src/signal.rs index 9393014dc..323748258 100644 --- a/litebox_broker_local/src/signal.rs +++ b/litebox_broker_local/src/signal.rs @@ -1,11 +1,13 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT license. +use litebox_broker_protocol::ObjectHandle; use litebox_broker_protocol::message::{ BrokerOperation, BrokerResult, SignalRequest, SignalResponse, }; -use litebox_broker_protocol::signal::{PendingSignal, SendSignalRequest, TakeSignalRequest}; -use litebox_broker_protocol::{ObjectHandle, ProcessId}; +use litebox_broker_protocol::signal::{ + SendSignalRequest, SignalEvent, SignalTarget, TakeSignalRequest, +}; use litebox_broker_transport::channel::LocalCallChannel; use crate::{BrokerLocal, BrokerLocalError, Result}; @@ -25,30 +27,28 @@ impl BrokerLocal { } } - /// Sends `signal` to process `process_id`, or only checks that it exists - /// if `signal` is zero. + /// Sends `signal` to the processes `target` selects, or only checks that + /// one exists if `signal` is zero. /// /// # Panics /// /// Panics if the broker reports an unrecoverable error or returns a protocol /// response that does not match the issued signal request. - pub fn send_signal(&self, process_id: ProcessId, signal: u32) -> Result<(), Channel::Error> { - match self.request_signal(SignalRequest::Send(SendSignalRequest { - process_id, - signal, - }))? { + pub fn send_signal(&self, target: SignalTarget, signal: u32) -> Result<(), Channel::Error> { + match self.request_signal(SignalRequest::Send(SendSignalRequest { target, signal }))? { SignalResponse::Sent => Ok(()), response => panic!("broker returned unexpected signal response: {response:?}"), } } - /// Takes this process's lowest-numbered pending signal. + /// Takes this process's lowest-numbered pending signal, or else its + /// pending child exit. /// /// # Panics /// /// Panics if the broker reports an unrecoverable error or returns a protocol /// response that does not match the issued signal request. - pub fn take_signal(&self, handle: ObjectHandle) -> Result { + pub fn take_signal(&self, handle: ObjectHandle) -> Result { match self.request_signal(SignalRequest::Take(TakeSignalRequest { handle }))? { SignalResponse::Take(signal) => Ok(signal), response => panic!("broker returned unexpected signal response: {response:?}"), diff --git a/litebox_broker_protocol/src/lib.rs b/litebox_broker_protocol/src/lib.rs index f110205c6..0c6ce29ed 100644 --- a/litebox_broker_protocol/src/lib.rs +++ b/litebox_broker_protocol/src/lib.rs @@ -22,6 +22,7 @@ pub mod fs; pub mod message; pub mod pipe; pub mod process; +pub mod process_group; pub mod random; pub mod readiness; pub mod shared_buffer; diff --git a/litebox_broker_protocol/src/message.rs b/litebox_broker_protocol/src/message.rs index c3a8054ab..48170535e 100644 --- a/litebox_broker_protocol/src/message.rs +++ b/litebox_broker_protocol/src/message.rs @@ -19,13 +19,14 @@ use crate::pipe::{ WritePipeResponse, }; use crate::process::{ - CreateThreadRequest, CreateThreadResponse, DuplicateObjectsToChildRequest, - ExitChildProcessRequest, ProcessExitStatus, ProcessStartupDescriptor, ProcessTermination, - StartChildProcessRequest, WriteChildMemoryRequest, + ChildExit, ChildSelector, CreateThreadRequest, CreateThreadResponse, + DuplicateObjectsToChildRequest, ExitChildProcessRequest, ProcessExitStatus, ProcessInfo, + ProcessStartupDescriptor, StartChildProcessRequest, WriteChildMemoryRequest, }; +use crate::process_group::SetProcessGroupRequest; use crate::readiness::ReadinessFlags; use crate::shared_buffer::SharedBufferSequence; -use crate::signal::{OpenSignalsResponse, PendingSignal, SendSignalRequest, TakeSignalRequest}; +use crate::signal::{OpenSignalsResponse, SendSignalRequest, SignalEvent, TakeSignalRequest}; use crate::socket::{ AcceptSocketRequest, AcceptSocketResponse, BindSocketRequest, BindSocketResponse, ConnectSocketRequest, ConnectSocketResponse, CreateSocketRequest, CreateSocketResponse, @@ -75,13 +76,18 @@ pub enum BrokerOperation { File(FileRequest), /// Start one pending child process. StartChildProcess(StartChildProcessRequest), - /// Read a process's termination status through a process handle without + /// Reap one exited child matching the selector, oldest first, without /// blocking. /// - /// The broker returns `WouldBlock` while the process is live. - GetProcessExitStatus(ObjectHandle), + /// The broker returns `WouldBlock` if only live children match, and + /// `UnknownObject` if none does. Each matching child's exit or removal is + /// then reported through the caller's signals. + ReapChild(ChildSelector), /// Record the exit of one pending child process that never started. ExitChildProcess(ExitChildProcessRequest), + /// Discard one pending child process that never started, as if it had + /// never been created, except that this process learns it was removed. + CancelChildProcess(ProcessId), /// Report this process's final termination status before its runner exits. /// /// The status is published once the runner exits, replacing the status the @@ -93,10 +99,22 @@ pub enum BrokerOperation { /// Each child applies the setting in effect when it terminates, so a change /// does not affect children that already terminated. SetChildReaping(bool), + /// Set whether this process adopts the orphaned children of its exiting + /// descendants. + /// + /// When a process exits, its children move to its nearest running ancestor + /// that adopts orphans, and are otherwise reaped when they exit. + SetOrphanAdoption(bool), + /// Read a process's place in the process tree. + /// + /// The broker returns `UnknownObject` if no such process exists. + GetProcessInfo(ProcessId), /// Duplicate object references into this process's pending child. DuplicateObjectsToChild(DuplicateObjectsToChildRequest), /// Write bytes into this process's pending child's memory image. WriteChildMemory(WriteChildMemoryRequest), + /// Process group and session request family. + ProcessGroup(ProcessGroupRequest), /// Timer object request family. Timer(TimerRequest), /// Signal request family. @@ -144,13 +162,17 @@ impl BrokerOperation { | Self::ExitThread(_) | Self::CloseObject(_) | Self::ExitChildProcess(_) + | Self::CancelChildProcess(_) | Self::ReportExitStatus(_) | Self::SetChildReaping(_) - | Self::GetProcessExitStatus(_) + | Self::SetOrphanAdoption(_) + | Self::GetProcessInfo(_) + | Self::ReapChild(_) | Self::CheckReadiness(_) | Self::GetStatusFlags(_) | Self::SetStatusFlags(_) | Self::Event(_) + | Self::ProcessGroup(_) | Self::Timer(_) | Self::Signal(_) | Self::Pipe(PipeRequest::Create(_)) @@ -242,6 +264,20 @@ pub enum TimerRequest { Read(ReadTimerRequest), } +/// Request about process groups and sessions. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ProcessGroupRequest { + /// Move a process into a process group. + Set(SetProcessGroupRequest), + /// Make a process the leader of a new session and of a new process group + /// in it. + /// + /// The process must be the caller or its pending child, or the broker + /// returns `UnknownObject`. The broker returns `PolicyDenied` if a process + /// group already has the process's ID. + CreateSession(ProcessId), +} + /// Request about signals sent between broker processes. #[derive(Clone, Debug, PartialEq, Eq)] pub enum SignalRequest { @@ -253,7 +289,8 @@ pub enum SignalRequest { Open, /// Send a signal to a process. Send(SendSignalRequest), - /// Take one of the caller's pending signals, lowest-numbered first. + /// Take one of the caller's pending signals, lowest-numbered first, then + /// its pending child exit, and then its pending child removal. Take(TakeSignalRequest), } @@ -326,19 +363,27 @@ pub enum BrokerResult { File(FileResponse), /// A pending child established its broker association. ProcessStarted, - /// Termination status of a child process. - ProcessExitStatus(ProcessTermination), + /// An exited child was reaped. + ChildReaped(ChildExit), /// A pending child's exit was recorded. ProcessExited, + /// A pending child was discarded. + ChildProcessCancelled, /// This process's final termination status was recorded. ExitStatusReported, /// This process's child-reaping setting was recorded. ChildReapingSet, + /// This process's orphan-adoption setting was recorded. + OrphanAdoptionSet, + /// A process's place in the process tree. + ProcessInfo(ProcessInfo), /// Object references were duplicated into a pending child, whose handles /// replaced the request's handles in its shared buffer. ObjectsDuplicated, /// Bytes were written into a pending child's memory image. ChildMemoryWritten, + /// Process group and session response family. + ProcessGroup(ProcessGroupResponse), /// Timer object response family. Timer(TimerResponse), /// Signal response family. @@ -380,6 +425,15 @@ pub enum TimerResponse { Read(ReadTimerResponse), } +/// Response to a process group request. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ProcessGroupResponse { + /// The process moved into the group. + Set, + /// The session was created. + CreateSession, +} + /// Response to a signal request. #[derive(Clone, Debug, PartialEq, Eq)] pub enum SignalResponse { @@ -388,7 +442,7 @@ pub enum SignalResponse { /// The signal was sent. Sent, /// Take operation response. - Take(PendingSignal), + Take(SignalEvent), } /// Broker-owned pipe object response. diff --git a/litebox_broker_protocol/src/process.rs b/litebox_broker_protocol/src/process.rs index 9d9cdf456..1c268cbf5 100644 --- a/litebox_broker_protocol/src/process.rs +++ b/litebox_broker_protocol/src/process.rs @@ -3,10 +3,11 @@ use alloc::vec::Vec; +use crate::process_group::ProcessGroupMembership; use crate::shared_buffer::{ MAX_SHARED_BUFFER_SEQUENCE_SLOTS, SHARED_BUFFER_SLOT_SIZE, SharedBufferSequence, }; -use crate::{ObjectHandle, ProcessId, ThreadId}; +use crate::{ProcessId, ThreadId}; /// Maximum size of one process bootstrap carried through the broker. pub const MAX_PROCESS_BOOTSTRAP_SIZE: u32 = 64 * 1024; @@ -49,27 +50,41 @@ pub enum ProcessExitStatus { Unknown, } -/// A terminated process's status as its handle observes it. +/// A child process's exit. #[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct ProcessTermination { - /// Termination status. +pub struct ChildExit { + /// Child that exited. + pub process_id: ProcessId, + /// Child's termination status. pub exit_status: ProcessExitStatus, - /// Whether the process was reaped when it terminated, so no wait reports - /// it. - pub reaped: bool, } -/// One newly created process and the creator's handle to it. -/// -/// The handle reports [`ReadinessFlags::READ`](crate::readiness::ReadinessFlags::READ) -/// once the process terminates. Closing it releases the process's retained -/// exit status. +/// Selects which of the caller's children a reap considers. #[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct CreatedProcess { - /// Broker-assigned process identity. - pub identity: ProcessIdentity, - /// Handle used to observe process termination. - pub handle: ObjectHandle, +pub enum ChildSelector { + /// Any child. + Any, + /// One child. + Process(ProcessId), + /// Any child in a process group. + ProcessGroup(ProcessId), +} + +/// A process's place in the process tree. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ProcessInfo { + /// Process that created this one, absent for a root process. + /// + /// The creator never changes, even after it exits. + pub creator: Option, + /// Process that reaps this one. + /// + /// The parent starts as the creator. When it exits, the nearest running + /// ancestor that adopts orphans becomes the parent, or none does and the + /// process is reaped when it exits. + pub parent: Option, + /// Process group and session. + pub membership: ProcessGroupMembership, } /// Selects whether thread creation extends the current process or creates a child process. @@ -87,7 +102,7 @@ pub enum CreateThreadResponse { /// A thread was created in the requesting process. Thread(ThreadId), /// A pending child process was created. - Process(CreatedProcess), + Process(ProcessIdentity), } /// Starts a pending child created earlier. diff --git a/litebox_broker_protocol/src/process_group.rs b/litebox_broker_protocol/src/process_group.rs new file mode 100644 index 000000000..59064fce9 --- /dev/null +++ b/litebox_broker_protocol/src/process_group.rs @@ -0,0 +1,34 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Process groups and sessions. +//! +//! Every process belongs to one process group, and every process group to one +//! session. Each is identified by the ID of the process that created it, which +//! need not still exist. A child process starts in its creator's group and +//! session, while a root process, which has no creator, leads its own. + +use crate::ProcessId; + +/// A process's group and session. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct ProcessGroupMembership { + /// Process group. + pub process_group: ProcessId, + /// Session. + pub session: ProcessId, +} + +/// Request to move a process into a process group. +/// +/// The target must be the caller or one of its children, or the broker returns +/// `UnknownObject`. The broker returns `PolicyDenied` if the target is in +/// another session than the caller or leads a session, or if `process_group` +/// is neither the target's ID nor an existing group in the caller's session. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct SetProcessGroupRequest { + /// Process to move. + pub process_id: ProcessId, + /// Group to join, which is created if it is `process_id`. + pub process_group: ProcessId, +} diff --git a/litebox_broker_protocol/src/signal.rs b/litebox_broker_protocol/src/signal.rs index 152b0a080..7ef8974ca 100644 --- a/litebox_broker_protocol/src/signal.rs +++ b/litebox_broker_protocol/src/signal.rs @@ -1,25 +1,41 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT license. -//! Signals sent between broker processes. +//! Signals sent between broker processes, and changes to their children +//! reported to parents. //! //! Signals are numbered from one to [`MAX_SIGNAL`]. The broker keeps at most //! one pending instance of each signal per process, so repeated signals -//! coalesce until the process takes them. +//! coalesce until the process takes them. Child events coalesce the same way: +//! a parent has at most one pending child exit, the earliest since it last +//! took one, and at most one pending child removal. +use crate::process::ChildExit; use crate::{ObjectHandle, ProcessId}; /// Largest signal number the broker delivers. pub const MAX_SIGNAL: u32 = 64; -/// Request to send a signal to a process. +/// Process or processes a signal is sent to. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum SignalTarget { + /// One process. + Process(ProcessId), + /// Every process in a process group. + ProcessGroup(ProcessId), + /// Every process except the caller and the root processes, which have no + /// creator. + All, +} + +/// Request to send a signal. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct SendSignalRequest { - /// Target process. + /// Target processes. /// - /// The broker returns `UnknownObject` if no such process exists. - pub process_id: ProcessId, - /// Signal number, or zero to only check that the target exists. + /// The broker returns `UnknownObject` if no process is targeted. + pub target: SignalTarget, + /// Signal number, or zero to only check that a target exists. /// /// The broker returns `UnsupportedOperation` for a number above /// [`MAX_SIGNAL`]. @@ -29,19 +45,34 @@ pub struct SendSignalRequest { /// Response to a request opening the caller's signals. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct OpenSignalsResponse { - /// Handle that becomes readable while signals are pending. + /// Handle that becomes readable while signals or child events are pending. pub handle: ObjectHandle, } -/// Request to take one of the caller's pending signals. +/// Request to take one of the caller's pending signals, its pending child +/// exit, or its pending child removal. /// -/// The broker returns `WouldBlock` when no signal is pending. +/// The broker returns `WouldBlock` when none is pending. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct TakeSignalRequest { /// Handle returned by the open request. pub handle: ObjectHandle, } +/// One event taken from the caller's signals. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum SignalEvent { + /// A signal another process sent. + Signal(PendingSignal), + /// The earliest child exit since the caller last took one. + ChildExited(ChildExit), + /// One or more children left the caller without exiting since it last + /// took this event, such as a child whose startup failed. + /// + /// A reap that found only such children live no longer finds them. + ChildRemoved, +} + /// A signal taken from the caller's pending signals. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub struct PendingSignal { diff --git a/litebox_broker_protocol/src/wire.rs b/litebox_broker_protocol/src/wire.rs index ade074523..f4c7fe77f 100644 --- a/litebox_broker_protocol/src/wire.rs +++ b/litebox_broker_protocol/src/wire.rs @@ -17,16 +17,18 @@ use alloc::vec::Vec; use thiserror::Error; +use crate::ProcessId; use crate::error::ErrorCode; use crate::message::{ BrokerHandshakeRequest, BrokerHandshakeResponse, BrokerNotification, BrokerOperation, BrokerRequest, BrokerResponse, BrokerResult, ReadinessNotification, }; use crate::process::{ - CreateThreadRequest, CreateThreadResponse, CreatedProcess, DuplicateObjectsToChildRequest, - ExitChildProcessRequest, ProcessExitStatus, ProcessIdentity, ProcessStartupDescriptor, - ProcessTermination, StartChildProcessRequest, WriteChildMemoryRequest, + ChildExit, ChildSelector, CreateThreadRequest, CreateThreadResponse, + DuplicateObjectsToChildRequest, ExitChildProcessRequest, ProcessExitStatus, ProcessIdentity, + ProcessInfo, ProcessStartupDescriptor, StartChildProcessRequest, WriteChildMemoryRequest, }; +use crate::process_group::ProcessGroupMembership; use crate::readiness::ReadinessFlags; use primitive::{Decoder, Encoder}; @@ -35,6 +37,7 @@ mod event; mod fs; mod pipe; mod primitive; +mod process_group; mod signal; mod socket; mod timer; @@ -50,7 +53,7 @@ const REQUEST_TAG_FILE: u8 = 8; const REQUEST_TAG_CREATE_THREAD: u8 = 9; const REQUEST_TAG_EXIT_THREAD: u8 = 10; const REQUEST_TAG_START_CHILD_PROCESS: u8 = 11; -const REQUEST_TAG_GET_PROCESS_EXIT_STATUS: u8 = 12; +const REQUEST_TAG_REAP_CHILD: u8 = 12; const REQUEST_TAG_EXIT_CHILD_PROCESS: u8 = 13; const REQUEST_TAG_REPORT_EXIT_STATUS: u8 = 14; const REQUEST_TAG_SET_CHILD_REAPING: u8 = 15; @@ -60,12 +63,19 @@ const REQUEST_TAG_GET_STATUS_FLAGS: u8 = 18; const REQUEST_TAG_SET_STATUS_FLAGS: u8 = 19; const REQUEST_TAG_SIGNAL: u8 = 20; const REQUEST_TAG_WRITE_CHILD_MEMORY: u8 = 21; +const REQUEST_TAG_PROCESS_GROUP: u8 = 22; +const REQUEST_TAG_CANCEL_CHILD_PROCESS: u8 = 23; +const REQUEST_TAG_SET_ORPHAN_ADOPTION: u8 = 24; +const REQUEST_TAG_GET_PROCESS_INFO: u8 = 25; const CREATE_THREAD_TAG_THREAD: u8 = 0; const CREATE_THREAD_TAG_PROCESS: u8 = 1; const PROCESS_EXIT_STATUS_TAG_EXITED: u8 = 0; const PROCESS_EXIT_STATUS_TAG_SIGNALED: u8 = 1; const PROCESS_EXIT_STATUS_TAG_UNKNOWN: u8 = 2; +const CHILD_SELECTOR_TAG_ANY: u8 = 0; +const CHILD_SELECTOR_TAG_PROCESS: u8 = 1; +const CHILD_SELECTOR_TAG_PROCESS_GROUP: u8 = 2; // Paired request and successful-response tags intentionally share values. const RESPONSE_TAG_NEGOTIATED: u8 = 0; @@ -79,7 +89,7 @@ const RESPONSE_TAG_FILE: u8 = 8; const RESPONSE_TAG_CREATE_THREAD: u8 = 9; const RESPONSE_TAG_THREAD_EXITED: u8 = 10; const RESPONSE_TAG_PROCESS_STARTED: u8 = 11; -const RESPONSE_TAG_PROCESS_EXIT_STATUS: u8 = 12; +const RESPONSE_TAG_CHILD_REAPED: u8 = 12; const RESPONSE_TAG_PROCESS_EXITED: u8 = 13; const RESPONSE_TAG_EXIT_STATUS_REPORTED: u8 = 14; const RESPONSE_TAG_CHILD_REAPING_SET: u8 = 15; @@ -89,6 +99,10 @@ const RESPONSE_TAG_STATUS_FLAGS: u8 = 18; const RESPONSE_TAG_STATUS_FLAGS_SET: u8 = 19; const RESPONSE_TAG_SIGNAL: u8 = 20; const RESPONSE_TAG_CHILD_MEMORY_WRITTEN: u8 = 21; +const RESPONSE_TAG_PROCESS_GROUP: u8 = 22; +const RESPONSE_TAG_CHILD_PROCESS_CANCELLED: u8 = 23; +const RESPONSE_TAG_ORPHAN_ADOPTION_SET: u8 = 24; +const RESPONSE_TAG_PROCESS_INFO: u8 = 25; // Reserve the top of the tag space for responses without paired requests. const RESPONSE_TAG_ERROR: u8 = 253; @@ -148,7 +162,7 @@ pub fn decode_handshake_request(frame: &[u8]) -> Result Result { + | REQUEST_TAG_WRITE_CHILD_MEMORY + | REQUEST_TAG_PROCESS_GROUP + | REQUEST_TAG_CANCEL_CHILD_PROCESS + | REQUEST_TAG_SET_ORPHAN_ADOPTION + | REQUEST_TAG_GET_PROCESS_INFO => { return Err(WireError::WrongMessagePhase); } _ => return Err(WireError::InvalidTag), @@ -241,10 +259,10 @@ pub fn encode_request(request: BrokerRequest) -> Vec { encoder.process_id(request.child_process_id); encoder.shared_buffer_sequence(request.startup.buffer); } - BrokerOperation::GetProcessExitStatus(handle) => { - encoder.u8(REQUEST_TAG_GET_PROCESS_EXIT_STATUS); + BrokerOperation::ReapChild(selector) => { + encoder.u8(REQUEST_TAG_REAP_CHILD); encoder.request_id(request_id); - encoder.handle(handle); + encode_child_selector(&mut encoder, selector); } BrokerOperation::ExitChildProcess(ExitChildProcessRequest { child_process_id, @@ -255,6 +273,11 @@ pub fn encode_request(request: BrokerRequest) -> Vec { encoder.process_id(child_process_id); encode_process_exit_status(&mut encoder, exit_status); } + BrokerOperation::CancelChildProcess(child_process_id) => { + encoder.u8(REQUEST_TAG_CANCEL_CHILD_PROCESS); + encoder.request_id(request_id); + encoder.process_id(child_process_id); + } BrokerOperation::ReportExitStatus(exit_status) => { encoder.u8(REQUEST_TAG_REPORT_EXIT_STATUS); encoder.request_id(request_id); @@ -265,6 +288,16 @@ pub fn encode_request(request: BrokerRequest) -> Vec { encoder.request_id(request_id); encoder.u8(u8::from(enabled)); } + BrokerOperation::SetOrphanAdoption(enabled) => { + encoder.u8(REQUEST_TAG_SET_ORPHAN_ADOPTION); + encoder.request_id(request_id); + encoder.u8(u8::from(enabled)); + } + BrokerOperation::GetProcessInfo(process_id) => { + encoder.u8(REQUEST_TAG_GET_PROCESS_INFO); + encoder.request_id(request_id); + encoder.process_id(process_id); + } BrokerOperation::DuplicateObjectsToChild(DuplicateObjectsToChildRequest { child_process_id, handles, @@ -285,6 +318,11 @@ pub fn encode_request(request: BrokerRequest) -> Vec { encoder.u64(offset); encoder.shared_buffer_sequence(data); } + BrokerOperation::ProcessGroup(request) => { + encoder.u8(REQUEST_TAG_PROCESS_GROUP); + encoder.request_id(request_id); + process_group::encode_process_group_request(&mut encoder, request); + } BrokerOperation::Timer(request) => { encoder.u8(REQUEST_TAG_TIMER); encoder.request_id(request_id); @@ -315,7 +353,7 @@ pub fn decode_request(frame: &[u8]) -> Result { | REQUEST_TAG_CREATE_THREAD | REQUEST_TAG_EXIT_THREAD | REQUEST_TAG_START_CHILD_PROCESS - | REQUEST_TAG_GET_PROCESS_EXIT_STATUS + | REQUEST_TAG_REAP_CHILD | REQUEST_TAG_EXIT_CHILD_PROCESS | REQUEST_TAG_REPORT_EXIT_STATUS | REQUEST_TAG_SET_CHILD_REAPING @@ -324,7 +362,11 @@ pub fn decode_request(frame: &[u8]) -> Result { | REQUEST_TAG_GET_STATUS_FLAGS | REQUEST_TAG_SET_STATUS_FLAGS | REQUEST_TAG_SIGNAL - | REQUEST_TAG_WRITE_CHILD_MEMORY => {} + | REQUEST_TAG_WRITE_CHILD_MEMORY + | REQUEST_TAG_PROCESS_GROUP + | REQUEST_TAG_CANCEL_CHILD_PROCESS + | REQUEST_TAG_SET_ORPHAN_ADOPTION + | REQUEST_TAG_GET_PROCESS_INFO => {} _ => return Err(WireError::InvalidTag), } let request_id = decoder.request_id()?; @@ -354,9 +396,7 @@ pub fn decode_request(frame: &[u8]) -> Result { }, }) } - REQUEST_TAG_GET_PROCESS_EXIT_STATUS => { - BrokerOperation::GetProcessExitStatus(decoder.handle()?) - } + REQUEST_TAG_REAP_CHILD => BrokerOperation::ReapChild(decode_child_selector(&mut decoder)?), REQUEST_TAG_EXIT_CHILD_PROCESS => { BrokerOperation::ExitChildProcess(ExitChildProcessRequest { child_process_id: decoder.process_id()?, @@ -366,11 +406,16 @@ pub fn decode_request(frame: &[u8]) -> Result { REQUEST_TAG_REPORT_EXIT_STATUS => { BrokerOperation::ReportExitStatus(decode_process_exit_status(&mut decoder)?) } - REQUEST_TAG_SET_CHILD_REAPING => BrokerOperation::SetChildReaping(match decoder.u8()? { - 0 => false, - 1 => true, - _ => return Err(WireError::InvalidTag), - }), + REQUEST_TAG_CANCEL_CHILD_PROCESS => { + BrokerOperation::CancelChildProcess(decoder.process_id()?) + } + REQUEST_TAG_SET_CHILD_REAPING => { + BrokerOperation::SetChildReaping(decode_bool(&mut decoder)?) + } + REQUEST_TAG_SET_ORPHAN_ADOPTION => { + BrokerOperation::SetOrphanAdoption(decode_bool(&mut decoder)?) + } + REQUEST_TAG_GET_PROCESS_INFO => BrokerOperation::GetProcessInfo(decoder.process_id()?), REQUEST_TAG_DUPLICATE_OBJECTS_TO_CHILD => { BrokerOperation::DuplicateObjectsToChild(DuplicateObjectsToChildRequest { child_process_id: decoder.process_id()?, @@ -384,6 +429,9 @@ pub fn decode_request(frame: &[u8]) -> Result { data: decoder.shared_buffer_sequence()?, }) } + REQUEST_TAG_PROCESS_GROUP => BrokerOperation::ProcessGroup( + process_group::decode_process_group_request(&mut decoder)?, + ), REQUEST_TAG_TIMER => BrokerOperation::Timer(timer::decode_timer_request(&mut decoder)?), REQUEST_TAG_SIGNAL => BrokerOperation::Signal(signal::decode_signal_request(&mut decoder)?), _ => unreachable!("active request tag was validated"), @@ -462,7 +510,7 @@ pub fn decode_handshake_response(frame: &[u8]) -> Result Result { + | RESPONSE_TAG_CHILD_MEMORY_WRITTEN + | RESPONSE_TAG_PROCESS_GROUP + | RESPONSE_TAG_CHILD_PROCESS_CANCELLED + | RESPONSE_TAG_ORPHAN_ADOPTION_SET + | RESPONSE_TAG_PROCESS_INFO => { return Err(WireError::WrongMessagePhase); } RESPONSE_TAG_VERSION_MISMATCH => BrokerHandshakeResponse::VersionMismatch { @@ -502,18 +554,13 @@ pub fn encode_response(response: BrokerResponse) -> Vec { encoder.u8(CREATE_THREAD_TAG_THREAD); encoder.thread_id(thread_id); } - CreateThreadResponse::Process(CreatedProcess { - identity: - ProcessIdentity { - process_id, - initial_thread_id, - }, - handle, + CreateThreadResponse::Process(ProcessIdentity { + process_id, + initial_thread_id, }) => { encoder.u8(CREATE_THREAD_TAG_PROCESS); encoder.process_id(process_id); encoder.thread_id(initial_thread_id); - encoder.handle(handle); } } } @@ -567,19 +614,19 @@ pub fn encode_response(response: BrokerResponse) -> Vec { encoder.u8(RESPONSE_TAG_PROCESS_STARTED); encoder.request_id(request_id); } - BrokerResult::ProcessExitStatus(ProcessTermination { - exit_status, - reaped, - }) => { - encoder.u8(RESPONSE_TAG_PROCESS_EXIT_STATUS); + BrokerResult::ChildReaped(child_exit) => { + encoder.u8(RESPONSE_TAG_CHILD_REAPED); encoder.request_id(request_id); - encode_process_exit_status(&mut encoder, exit_status); - encoder.u8(u8::from(reaped)); + encode_child_exit(&mut encoder, child_exit); } BrokerResult::ProcessExited => { encoder.u8(RESPONSE_TAG_PROCESS_EXITED); encoder.request_id(request_id); } + BrokerResult::ChildProcessCancelled => { + encoder.u8(RESPONSE_TAG_CHILD_PROCESS_CANCELLED); + encoder.request_id(request_id); + } BrokerResult::ExitStatusReported => { encoder.u8(RESPONSE_TAG_EXIT_STATUS_REPORTED); encoder.request_id(request_id); @@ -588,6 +635,15 @@ pub fn encode_response(response: BrokerResponse) -> Vec { encoder.u8(RESPONSE_TAG_CHILD_REAPING_SET); encoder.request_id(request_id); } + BrokerResult::OrphanAdoptionSet => { + encoder.u8(RESPONSE_TAG_ORPHAN_ADOPTION_SET); + encoder.request_id(request_id); + } + BrokerResult::ProcessInfo(info) => { + encoder.u8(RESPONSE_TAG_PROCESS_INFO); + encoder.request_id(request_id); + encode_process_info(&mut encoder, info); + } BrokerResult::ObjectsDuplicated => { encoder.u8(RESPONSE_TAG_OBJECTS_DUPLICATED); encoder.request_id(request_id); @@ -596,6 +652,11 @@ pub fn encode_response(response: BrokerResponse) -> Vec { encoder.u8(RESPONSE_TAG_CHILD_MEMORY_WRITTEN); encoder.request_id(request_id); } + BrokerResult::ProcessGroup(response) => { + encoder.u8(RESPONSE_TAG_PROCESS_GROUP); + encoder.request_id(request_id); + process_group::encode_process_group_response(&mut encoder, response); + } BrokerResult::Timer(response) => { encoder.u8(RESPONSE_TAG_TIMER); encoder.request_id(request_id); @@ -634,7 +695,7 @@ pub fn decode_response(frame: &[u8]) -> Result { | RESPONSE_TAG_CREATE_THREAD | RESPONSE_TAG_THREAD_EXITED | RESPONSE_TAG_PROCESS_STARTED - | RESPONSE_TAG_PROCESS_EXIT_STATUS + | RESPONSE_TAG_CHILD_REAPED | RESPONSE_TAG_PROCESS_EXITED | RESPONSE_TAG_EXIT_STATUS_REPORTED | RESPONSE_TAG_CHILD_REAPING_SET @@ -643,7 +704,11 @@ pub fn decode_response(frame: &[u8]) -> Result { | RESPONSE_TAG_STATUS_FLAGS | RESPONSE_TAG_STATUS_FLAGS_SET | RESPONSE_TAG_SIGNAL - | RESPONSE_TAG_CHILD_MEMORY_WRITTEN => {} + | RESPONSE_TAG_CHILD_MEMORY_WRITTEN + | RESPONSE_TAG_PROCESS_GROUP + | RESPONSE_TAG_CHILD_PROCESS_CANCELLED + | RESPONSE_TAG_ORPHAN_ADOPTION_SET + | RESPONSE_TAG_PROCESS_INFO => {} _ => return Err(WireError::InvalidTag), } let request_id = decoder.request_id()?; @@ -654,12 +719,9 @@ pub fn decode_response(frame: &[u8]) -> Result { RESPONSE_TAG_ERROR => BrokerResult::Error(decode_error_code(&mut decoder)?), RESPONSE_TAG_CREATE_THREAD => BrokerResult::CreateThread(match decoder.u8()? { CREATE_THREAD_TAG_THREAD => CreateThreadResponse::Thread(decoder.thread_id()?), - CREATE_THREAD_TAG_PROCESS => CreateThreadResponse::Process(CreatedProcess { - identity: ProcessIdentity { - process_id: decoder.process_id()?, - initial_thread_id: decoder.thread_id()?, - }, - handle: decoder.handle()?, + CREATE_THREAD_TAG_PROCESS => CreateThreadResponse::Process(ProcessIdentity { + process_id: decoder.process_id()?, + initial_thread_id: decoder.thread_id()?, }), _ => return Err(WireError::InvalidTag), }), @@ -673,19 +735,18 @@ pub fn decode_response(frame: &[u8]) -> Result { RESPONSE_TAG_RANDOM_FILLED => BrokerResult::RandomFilled, RESPONSE_TAG_FILE => BrokerResult::File(fs::decode_fs_response(&mut decoder)?), RESPONSE_TAG_PROCESS_STARTED => BrokerResult::ProcessStarted, - RESPONSE_TAG_PROCESS_EXIT_STATUS => BrokerResult::ProcessExitStatus(ProcessTermination { - exit_status: decode_process_exit_status(&mut decoder)?, - reaped: match decoder.u8()? { - 0 => false, - 1 => true, - _ => return Err(WireError::InvalidTag), - }, - }), + RESPONSE_TAG_CHILD_REAPED => BrokerResult::ChildReaped(decode_child_exit(&mut decoder)?), RESPONSE_TAG_PROCESS_EXITED => BrokerResult::ProcessExited, + RESPONSE_TAG_CHILD_PROCESS_CANCELLED => BrokerResult::ChildProcessCancelled, RESPONSE_TAG_EXIT_STATUS_REPORTED => BrokerResult::ExitStatusReported, RESPONSE_TAG_CHILD_REAPING_SET => BrokerResult::ChildReapingSet, + RESPONSE_TAG_ORPHAN_ADOPTION_SET => BrokerResult::OrphanAdoptionSet, + RESPONSE_TAG_PROCESS_INFO => BrokerResult::ProcessInfo(decode_process_info(&mut decoder)?), RESPONSE_TAG_OBJECTS_DUPLICATED => BrokerResult::ObjectsDuplicated, RESPONSE_TAG_CHILD_MEMORY_WRITTEN => BrokerResult::ChildMemoryWritten, + RESPONSE_TAG_PROCESS_GROUP => { + BrokerResult::ProcessGroup(process_group::decode_process_group_response(&mut decoder)?) + } RESPONSE_TAG_TIMER => BrokerResult::Timer(timer::decode_timer_response(&mut decoder)?), RESPONSE_TAG_SIGNAL => BrokerResult::Signal(signal::decode_signal_response(&mut decoder)?), _ => unreachable!("active response tag was validated"), @@ -721,6 +782,85 @@ fn decode_process_exit_status(decoder: &mut Decoder<'_>) -> Result) -> Result { + Ok(ChildExit { + process_id: decoder.process_id()?, + exit_status: decode_process_exit_status(decoder)?, + }) +} + +fn encode_child_selector(encoder: &mut Encoder, selector: ChildSelector) { + match selector { + ChildSelector::Any => encoder.u8(CHILD_SELECTOR_TAG_ANY), + ChildSelector::Process(process_id) => { + encoder.u8(CHILD_SELECTOR_TAG_PROCESS); + encoder.process_id(process_id); + } + ChildSelector::ProcessGroup(process_group) => { + encoder.u8(CHILD_SELECTOR_TAG_PROCESS_GROUP); + encoder.process_id(process_group); + } + } +} + +fn decode_child_selector(decoder: &mut Decoder<'_>) -> Result { + Ok(match decoder.u8()? { + CHILD_SELECTOR_TAG_ANY => ChildSelector::Any, + CHILD_SELECTOR_TAG_PROCESS => ChildSelector::Process(decoder.process_id()?), + CHILD_SELECTOR_TAG_PROCESS_GROUP => ChildSelector::ProcessGroup(decoder.process_id()?), + _ => return Err(WireError::InvalidTag), + }) +} + +fn encode_process_info(encoder: &mut Encoder, info: ProcessInfo) { + encode_optional_process_id(encoder, info.creator); + encode_optional_process_id(encoder, info.parent); + encoder.process_id(info.membership.process_group); + encoder.process_id(info.membership.session); +} + +fn decode_process_info(decoder: &mut Decoder<'_>) -> Result { + Ok(ProcessInfo { + creator: decode_optional_process_id(decoder)?, + parent: decode_optional_process_id(decoder)?, + membership: ProcessGroupMembership { + process_group: decoder.process_id()?, + session: decoder.process_id()?, + }, + }) +} + +fn encode_optional_process_id(encoder: &mut Encoder, process_id: Option) { + match process_id { + Some(process_id) => { + encoder.u8(1); + encoder.process_id(process_id); + } + None => encoder.u8(0), + } +} + +fn decode_optional_process_id(decoder: &mut Decoder<'_>) -> Result, WireError> { + match decoder.u8()? { + 0 => Ok(None), + 1 => Ok(Some(decoder.process_id()?)), + _ => Err(WireError::InvalidTag), + } +} + +fn decode_bool(decoder: &mut Decoder<'_>) -> Result { + match decoder.u8()? { + 0 => Ok(false), + 1 => Ok(true), + _ => Err(WireError::InvalidTag), + } +} + fn encode_error_code(encoder: &mut Encoder, error: ErrorCode) { encoder.u16(match error { ErrorCode::UnsupportedVersion => 1, @@ -807,22 +947,28 @@ mod tests { }; use crate::message::{ EventRequest, EventResponse, FileRequest, FileResponse, PipeRequest, PipeResponse, - SignalRequest, SignalResponse, SocketRequest, SocketResponse, TimerRequest, TimerResponse, + ProcessGroupRequest, ProcessGroupResponse, SignalRequest, SignalResponse, SocketRequest, + SocketResponse, TimerRequest, TimerResponse, }; use crate::pipe::{ CreatePipeRequest, CreatePipeResponse, ReadPipeRequest, ReadPipeResponse, WritePipeRequest, WritePipeResponse, }; use crate::process::{ - CreateThreadRequest, CreateThreadResponse, CreatedProcess, DuplicateObjectsToChildRequest, - ExitChildProcessRequest, ProcessExitStatus, ProcessIdentity, ProcessStartupDescriptor, - ProcessTermination, StartChildProcessRequest, WriteChildMemoryRequest, + ChildExit, ChildSelector, CreateThreadRequest, CreateThreadResponse, + DuplicateObjectsToChildRequest, ExitChildProcessRequest, ProcessExitStatus, + ProcessIdentity, ProcessInfo, ProcessStartupDescriptor, StartChildProcessRequest, + WriteChildMemoryRequest, }; + use crate::process_group::{ProcessGroupMembership, SetProcessGroupRequest}; use crate::shared_buffer::{ MAX_SHARED_BUFFER_SEQUENCE_SLOTS, SHARED_BUFFER_SLOT_SIZE, SharedBufferSequence, SharedBufferSlotIndex, }; - use crate::signal::{OpenSignalsResponse, PendingSignal, SendSignalRequest, TakeSignalRequest}; + use crate::signal::{ + OpenSignalsResponse, PendingSignal, SendSignalRequest, SignalEvent, SignalTarget, + TakeSignalRequest, + }; use crate::socket::{ AcceptSocketRequest, AcceptSocketResponse, AddressFamily, BindSocketRequest, BindSocketResponse, ConnectSocketRequest, ConnectSocketResponse, CreateSocketRequest, @@ -879,7 +1025,7 @@ mod tests { RESPONSE_TAG_CREATE_THREAD, RESPONSE_TAG_THREAD_EXITED, RESPONSE_TAG_PROCESS_STARTED, - RESPONSE_TAG_PROCESS_EXIT_STATUS, + RESPONSE_TAG_CHILD_REAPED, RESPONSE_TAG_PROCESS_EXITED, RESPONSE_TAG_EXIT_STATUS_REPORTED, RESPONSE_TAG_CHILD_REAPING_SET, @@ -889,6 +1035,10 @@ mod tests { RESPONSE_TAG_STATUS_FLAGS_SET, RESPONSE_TAG_SIGNAL, RESPONSE_TAG_CHILD_MEMORY_WRITTEN, + RESPONSE_TAG_PROCESS_GROUP, + RESPONSE_TAG_CHILD_PROCESS_CANCELLED, + RESPONSE_TAG_ORPHAN_ADOPTION_SET, + RESPONSE_TAG_PROCESS_INFO, ], [ REQUEST_TAG_NEGOTIATE, @@ -902,7 +1052,7 @@ mod tests { REQUEST_TAG_CREATE_THREAD, REQUEST_TAG_EXIT_THREAD, REQUEST_TAG_START_CHILD_PROCESS, - REQUEST_TAG_GET_PROCESS_EXIT_STATUS, + REQUEST_TAG_REAP_CHILD, REQUEST_TAG_EXIT_CHILD_PROCESS, REQUEST_TAG_REPORT_EXIT_STATUS, REQUEST_TAG_SET_CHILD_REAPING, @@ -912,6 +1062,10 @@ mod tests { REQUEST_TAG_SET_STATUS_FLAGS, REQUEST_TAG_SIGNAL, REQUEST_TAG_WRITE_CHILD_MEMORY, + REQUEST_TAG_PROCESS_GROUP, + REQUEST_TAG_CANCEL_CHILD_PROCESS, + REQUEST_TAG_SET_ORPHAN_ADOPTION, + REQUEST_TAG_GET_PROCESS_INFO, ] ); assert_eq!( @@ -991,9 +1145,22 @@ mod tests { BrokerOperation::Timer(TimerRequest::Read(ReadTimerRequest { handle })), BrokerOperation::Signal(SignalRequest::Open), BrokerOperation::Signal(SignalRequest::Send(SendSignalRequest { - process_id: process_id(u32::MAX), + target: SignalTarget::Process(process_id(u32::MAX)), signal: 9, })), + BrokerOperation::Signal(SignalRequest::Send(SendSignalRequest { + target: SignalTarget::ProcessGroup(process_id(7)), + signal: 0, + })), + BrokerOperation::Signal(SignalRequest::Send(SendSignalRequest { + target: SignalTarget::All, + signal: 64, + })), + BrokerOperation::ProcessGroup(ProcessGroupRequest::Set(SetProcessGroupRequest { + process_id: process_id(3), + process_group: process_id(u32::MAX), + })), + BrokerOperation::ProcessGroup(ProcessGroupRequest::CreateSession(process_id(5))), BrokerOperation::Signal(SignalRequest::Take(TakeSignalRequest { handle })), BrokerOperation::Pipe(PipeRequest::Create(CreatePipeRequest { capacity: 4096, @@ -1186,7 +1353,13 @@ mod tests { buffer: largest_sequence, }, }), - BrokerOperation::GetProcessExitStatus(ObjectHandle(u64::MAX)), + BrokerOperation::ReapChild(ChildSelector::Any), + BrokerOperation::ReapChild(ChildSelector::Process(process_id(u32::MAX))), + BrokerOperation::ReapChild(ChildSelector::ProcessGroup(process_id(7))), + BrokerOperation::CancelChildProcess(process_id(u32::MAX)), + BrokerOperation::SetOrphanAdoption(false), + BrokerOperation::SetOrphanAdoption(true), + BrokerOperation::GetProcessInfo(process_id(u32::MAX)), BrokerOperation::ExitChildProcess(ExitChildProcessRequest { child_process_id: process_id(u32::MAX), exit_status: ProcessExitStatus::Exited { code: u32::MAX }, @@ -1417,12 +1590,9 @@ mod tests { let handle = ObjectHandle(13); let results = [ BrokerResult::CreateThread(CreateThreadResponse::Thread(thread_id(17))), - BrokerResult::CreateThread(CreateThreadResponse::Process(CreatedProcess { - identity: ProcessIdentity { - process_id: process_id(19), - initial_thread_id: thread_id(21), - }, - handle: ObjectHandle(u64::MAX), + BrokerResult::CreateThread(CreateThreadResponse::Process(ProcessIdentity { + process_id: process_id(19), + initial_thread_id: thread_id(21), })), BrokerResult::ThreadExited, BrokerResult::ObjectClosed, @@ -1460,10 +1630,17 @@ mod tests { })), BrokerResult::Signal(SignalResponse::Open(OpenSignalsResponse { handle })), BrokerResult::Signal(SignalResponse::Sent), - BrokerResult::Signal(SignalResponse::Take(PendingSignal { + BrokerResult::Signal(SignalResponse::Take(SignalEvent::Signal(PendingSignal { signal: 64, sender: process_id(u32::MAX), - })), + }))), + BrokerResult::Signal(SignalResponse::Take(SignalEvent::ChildExited(ChildExit { + process_id: process_id(u32::MAX), + exit_status: ProcessExitStatus::Signaled { signal: 9 }, + }))), + BrokerResult::Signal(SignalResponse::Take(SignalEvent::ChildRemoved)), + BrokerResult::ProcessGroup(ProcessGroupResponse::Set), + BrokerResult::ProcessGroup(ProcessGroupResponse::CreateSession), BrokerResult::Pipe(PipeResponse::Create(CreatePipeResponse { read_handle: handle, write_handle: ObjectHandle(14), @@ -1581,21 +1758,39 @@ mod tests { BrokerResult::File(FileResponse::Rmdir), BrokerResult::File(FileResponse::Failed(FileError::Io)), BrokerResult::ProcessStarted, - BrokerResult::ProcessExitStatus(ProcessTermination { + BrokerResult::ChildReaped(ChildExit { + process_id: process_id(u32::MAX), exit_status: ProcessExitStatus::Exited { code: u32::MAX }, - reaped: false, }), - BrokerResult::ProcessExitStatus(ProcessTermination { + BrokerResult::ChildReaped(ChildExit { + process_id: process_id(1), exit_status: ProcessExitStatus::Signaled { signal: 11 }, - reaped: true, }), - BrokerResult::ProcessExitStatus(ProcessTermination { + BrokerResult::ChildReaped(ChildExit { + process_id: process_id(1), exit_status: ProcessExitStatus::Unknown, - reaped: false, }), BrokerResult::ProcessExited, + BrokerResult::ChildProcessCancelled, BrokerResult::ExitStatusReported, BrokerResult::ChildReapingSet, + BrokerResult::OrphanAdoptionSet, + BrokerResult::ProcessInfo(ProcessInfo { + creator: None, + parent: None, + membership: ProcessGroupMembership { + process_group: process_id(1), + session: process_id(1), + }, + }), + BrokerResult::ProcessInfo(ProcessInfo { + creator: Some(process_id(u32::MAX)), + parent: Some(process_id(1)), + membership: ProcessGroupMembership { + process_group: process_id(u32::MAX), + session: process_id(3), + }, + }), BrokerResult::ObjectsDuplicated, BrokerResult::ChildMemoryWritten, BrokerResult::Error(ErrorCode::PolicyDenied), @@ -1649,13 +1844,12 @@ mod tests { for process_id in [ProcessId(0), ProcessId(u32::MAX)] { let response = BrokerResponse { request_id: TEST_REQUEST_ID, - result: BrokerResult::CreateThread(CreateThreadResponse::Process(CreatedProcess { - identity: ProcessIdentity { + result: BrokerResult::CreateThread(CreateThreadResponse::Process( + ProcessIdentity { process_id, initial_thread_id: ThreadId(0), }, - handle: ObjectHandle(0), - })), + )), }; assert_eq!( decode_response(&encode_response(response.clone())).unwrap(), @@ -1775,7 +1969,7 @@ mod tests { let truncated_signal_send = encode_request(BrokerRequest { request_id: TEST_REQUEST_ID, operation: BrokerOperation::Signal(SignalRequest::Send(SendSignalRequest { - process_id: process_id(1), + target: SignalTarget::Process(process_id(1)), signal: 10, })), }); @@ -1783,6 +1977,42 @@ mod tests { decode_request(&truncated_signal_send[..truncated_signal_send.len() - 1]), Err(WireError::TruncatedFrame) ); + let mut unknown_signal_target = encode_request(BrokerRequest { + request_id: TEST_REQUEST_ID, + operation: BrokerOperation::Signal(SignalRequest::Send(SendSignalRequest { + target: SignalTarget::All, + signal: 10, + })), + }); + unknown_signal_target[10] = 0xff; + assert_eq!( + decode_request(&unknown_signal_target), + Err(WireError::InvalidTag) + ); + let mut unknown_process_group_request = encode_request(BrokerRequest { + request_id: TEST_REQUEST_ID, + operation: BrokerOperation::ProcessGroup(ProcessGroupRequest::CreateSession( + process_id(1), + )), + }); + unknown_process_group_request[9] = 0xff; + assert_eq!( + decode_request(&unknown_process_group_request), + Err(WireError::InvalidTag) + ); + let truncated_process_group_set = encode_request(BrokerRequest { + request_id: TEST_REQUEST_ID, + operation: BrokerOperation::ProcessGroup(ProcessGroupRequest::Set( + SetProcessGroupRequest { + process_id: process_id(1), + process_group: process_id(2), + }, + )), + }); + assert_eq!( + decode_request(&truncated_process_group_set[..truncated_process_group_set.len() - 1]), + Err(WireError::TruncatedFrame) + ); let mut unknown_status_flag = encode_request(BrokerRequest { request_id: TEST_REQUEST_ID, operation: BrokerOperation::SetStatusFlags(SetStatusFlagsRequest { @@ -1826,6 +2056,24 @@ mod tests { decode_request(&invalid_child_reaping), Err(WireError::InvalidTag) ); + let mut invalid_orphan_adoption = encode_request(BrokerRequest { + request_id: TEST_REQUEST_ID, + operation: BrokerOperation::SetOrphanAdoption(true), + }); + invalid_orphan_adoption[9] = 2; + assert_eq!( + decode_request(&invalid_orphan_adoption), + Err(WireError::InvalidTag) + ); + let mut unknown_child_selector = encode_request(BrokerRequest { + request_id: TEST_REQUEST_ID, + operation: BrokerOperation::ReapChild(ChildSelector::Any), + }); + unknown_child_selector[9] = 0xff; + assert_eq!( + decode_request(&unknown_child_selector), + Err(WireError::InvalidTag) + ); let mut frame = encode_request(BrokerRequest { request_id: TEST_REQUEST_ID, operation: BrokerOperation::Event(EventRequest::Create(CreateEventRequest { @@ -2282,25 +2530,32 @@ mod tests { let mut invalid_exit_status = encode_response(BrokerResponse { request_id: TEST_REQUEST_ID, - result: BrokerResult::ProcessExitStatus(ProcessTermination { + result: BrokerResult::ChildReaped(ChildExit { + process_id: process_id(1), exit_status: ProcessExitStatus::Unknown, - reaped: false, }), }); - invalid_exit_status[9] = 0xff; + invalid_exit_status[13] = 0xff; assert_eq!( decode_response(&invalid_exit_status), Err(WireError::InvalidTag) ); - let mut invalid_reaped = encode_response(BrokerResponse { + let mut invalid_creator = encode_response(BrokerResponse { request_id: TEST_REQUEST_ID, - result: BrokerResult::ProcessExitStatus(ProcessTermination { - exit_status: ProcessExitStatus::Unknown, - reaped: false, + result: BrokerResult::ProcessInfo(ProcessInfo { + creator: None, + parent: None, + membership: ProcessGroupMembership { + process_group: process_id(1), + session: process_id(1), + }, }), }); - invalid_reaped[10] = 2; - assert_eq!(decode_response(&invalid_reaped), Err(WireError::InvalidTag)); + invalid_creator[9] = 2; + assert_eq!( + decode_response(&invalid_creator), + Err(WireError::InvalidTag) + ); let mut unknown_timer_response = encode_response(BrokerResponse { request_id: TEST_REQUEST_ID, @@ -2321,6 +2576,20 @@ mod tests { decode_response(&unknown_signal_response), Err(WireError::InvalidTag) ); + let mut unknown_signal_event = encode_response(BrokerResponse { + request_id: TEST_REQUEST_ID, + result: BrokerResult::Signal(SignalResponse::Take(SignalEvent::Signal( + PendingSignal { + signal: 1, + sender: process_id(1), + }, + ))), + }); + unknown_signal_event[10] = 0xff; + assert_eq!( + decode_response(&unknown_signal_event), + Err(WireError::InvalidTag) + ); let mut frame = encode_response(BrokerResponse { request_id: TEST_REQUEST_ID, diff --git a/litebox_broker_protocol/src/wire/process_group.rs b/litebox_broker_protocol/src/wire/process_group.rs new file mode 100644 index 000000000..7873aaeac --- /dev/null +++ b/litebox_broker_protocol/src/wire/process_group.rs @@ -0,0 +1,62 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use crate::message::{ProcessGroupRequest, ProcessGroupResponse}; +use crate::process_group::SetProcessGroupRequest; + +use super::WireError; +use super::primitive::{Decoder, Encoder}; + +const PROCESS_GROUP_REQUEST_TAG_SET: u8 = 0; +const PROCESS_GROUP_REQUEST_TAG_CREATE_SESSION: u8 = 1; + +const PROCESS_GROUP_RESPONSE_TAG_SET: u8 = 0; +const PROCESS_GROUP_RESPONSE_TAG_CREATE_SESSION: u8 = 1; + +pub(super) fn encode_process_group_request(encoder: &mut Encoder, request: ProcessGroupRequest) { + match request { + ProcessGroupRequest::Set(request) => { + encoder.u8(PROCESS_GROUP_REQUEST_TAG_SET); + encoder.process_id(request.process_id); + encoder.process_id(request.process_group); + } + ProcessGroupRequest::CreateSession(process_id) => { + encoder.u8(PROCESS_GROUP_REQUEST_TAG_CREATE_SESSION); + encoder.process_id(process_id); + } + } +} + +pub(super) fn decode_process_group_request( + decoder: &mut Decoder<'_>, +) -> Result { + Ok(match decoder.u8()? { + PROCESS_GROUP_REQUEST_TAG_SET => ProcessGroupRequest::Set(SetProcessGroupRequest { + process_id: decoder.process_id()?, + process_group: decoder.process_id()?, + }), + PROCESS_GROUP_REQUEST_TAG_CREATE_SESSION => { + ProcessGroupRequest::CreateSession(decoder.process_id()?) + } + _ => return Err(WireError::InvalidTag), + }) +} + +pub(super) fn encode_process_group_response(encoder: &mut Encoder, response: ProcessGroupResponse) { + match response { + ProcessGroupResponse::Set => encoder.u8(PROCESS_GROUP_RESPONSE_TAG_SET), + ProcessGroupResponse::CreateSession => { + encoder.u8(PROCESS_GROUP_RESPONSE_TAG_CREATE_SESSION); + } + } +} + +pub(super) fn decode_process_group_response( + decoder: &mut Decoder<'_>, +) -> Result { + Ok(match decoder.u8()? { + PROCESS_GROUP_RESPONSE_TAG_SET => ProcessGroupResponse::Set, + PROCESS_GROUP_RESPONSE_TAG_CREATE_SESSION => ProcessGroupResponse::CreateSession, + _ => return Err(WireError::InvalidTag), + }) +} diff --git a/litebox_broker_protocol/src/wire/signal.rs b/litebox_broker_protocol/src/wire/signal.rs index 2b4bafbfd..036000019 100644 --- a/litebox_broker_protocol/src/wire/signal.rs +++ b/litebox_broker_protocol/src/wire/signal.rs @@ -2,25 +2,36 @@ // Licensed under the MIT license. use crate::message::{SignalRequest, SignalResponse}; -use crate::signal::{OpenSignalsResponse, PendingSignal, SendSignalRequest, TakeSignalRequest}; +use crate::signal::{ + OpenSignalsResponse, PendingSignal, SendSignalRequest, SignalEvent, SignalTarget, + TakeSignalRequest, +}; -use super::WireError; use super::primitive::{Decoder, Encoder}; +use super::{WireError, decode_child_exit, encode_child_exit}; const SIGNAL_REQUEST_TAG_OPEN: u8 = 0; const SIGNAL_REQUEST_TAG_SEND: u8 = 1; const SIGNAL_REQUEST_TAG_TAKE: u8 = 2; +const SIGNAL_TARGET_TAG_PROCESS: u8 = 0; +const SIGNAL_TARGET_TAG_PROCESS_GROUP: u8 = 1; +const SIGNAL_TARGET_TAG_ALL: u8 = 2; + const SIGNAL_RESPONSE_TAG_OPEN: u8 = 0; const SIGNAL_RESPONSE_TAG_SENT: u8 = 1; const SIGNAL_RESPONSE_TAG_TAKE: u8 = 2; +const SIGNAL_EVENT_TAG_SIGNAL: u8 = 0; +const SIGNAL_EVENT_TAG_CHILD_EXITED: u8 = 1; +const SIGNAL_EVENT_TAG_CHILD_REMOVED: u8 = 2; + pub(super) fn encode_signal_request(encoder: &mut Encoder, request: SignalRequest) { match request { SignalRequest::Open => encoder.u8(SIGNAL_REQUEST_TAG_OPEN), SignalRequest::Send(request) => { encoder.u8(SIGNAL_REQUEST_TAG_SEND); - encoder.process_id(request.process_id); + encode_signal_target(encoder, request.target); encoder.u32(request.signal); } SignalRequest::Take(request) => { @@ -34,7 +45,7 @@ pub(super) fn decode_signal_request(decoder: &mut Decoder<'_>) -> Result SignalRequest::Open, SIGNAL_REQUEST_TAG_SEND => SignalRequest::Send(SendSignalRequest { - process_id: decoder.process_id()?, + target: decode_signal_target(decoder)?, signal: decoder.u32()?, }), SIGNAL_REQUEST_TAG_TAKE => SignalRequest::Take(TakeSignalRequest { @@ -44,6 +55,29 @@ pub(super) fn decode_signal_request(decoder: &mut Decoder<'_>) -> Result { + encoder.u8(SIGNAL_TARGET_TAG_PROCESS); + encoder.process_id(process_id); + } + SignalTarget::ProcessGroup(process_group) => { + encoder.u8(SIGNAL_TARGET_TAG_PROCESS_GROUP); + encoder.process_id(process_group); + } + SignalTarget::All => encoder.u8(SIGNAL_TARGET_TAG_ALL), + } +} + +fn decode_signal_target(decoder: &mut Decoder<'_>) -> Result { + Ok(match decoder.u8()? { + SIGNAL_TARGET_TAG_PROCESS => SignalTarget::Process(decoder.process_id()?), + SIGNAL_TARGET_TAG_PROCESS_GROUP => SignalTarget::ProcessGroup(decoder.process_id()?), + SIGNAL_TARGET_TAG_ALL => SignalTarget::All, + _ => return Err(WireError::InvalidTag), + }) +} + pub(super) fn encode_signal_response(encoder: &mut Encoder, response: SignalResponse) { match response { SignalResponse::Open(response) => { @@ -51,10 +85,20 @@ pub(super) fn encode_signal_response(encoder: &mut Encoder, response: SignalResp encoder.handle(response.handle); } SignalResponse::Sent => encoder.u8(SIGNAL_RESPONSE_TAG_SENT), - SignalResponse::Take(response) => { + SignalResponse::Take(event) => { encoder.u8(SIGNAL_RESPONSE_TAG_TAKE); - encoder.u32(response.signal); - encoder.process_id(response.sender); + match event { + SignalEvent::Signal(signal) => { + encoder.u8(SIGNAL_EVENT_TAG_SIGNAL); + encoder.u32(signal.signal); + encoder.process_id(signal.sender); + } + SignalEvent::ChildExited(child_exit) => { + encoder.u8(SIGNAL_EVENT_TAG_CHILD_EXITED); + encode_child_exit(encoder, child_exit); + } + SignalEvent::ChildRemoved => encoder.u8(SIGNAL_EVENT_TAG_CHILD_REMOVED), + } } } } @@ -67,9 +111,14 @@ pub(super) fn decode_signal_response( handle: decoder.handle()?, }), SIGNAL_RESPONSE_TAG_SENT => SignalResponse::Sent, - SIGNAL_RESPONSE_TAG_TAKE => SignalResponse::Take(PendingSignal { - signal: decoder.u32()?, - sender: decoder.process_id()?, + SIGNAL_RESPONSE_TAG_TAKE => SignalResponse::Take(match decoder.u8()? { + SIGNAL_EVENT_TAG_SIGNAL => SignalEvent::Signal(PendingSignal { + signal: decoder.u32()?, + sender: decoder.process_id()?, + }), + SIGNAL_EVENT_TAG_CHILD_EXITED => SignalEvent::ChildExited(decode_child_exit(decoder)?), + SIGNAL_EVENT_TAG_CHILD_REMOVED => SignalEvent::ChildRemoved, + _ => return Err(WireError::InvalidTag), }), _ => return Err(WireError::InvalidTag), }) diff --git a/litebox_broker_userland/tests/userland_broker.rs b/litebox_broker_userland/tests/userland_broker.rs index 6669174de..35bf81158 100644 --- a/litebox_broker_userland/tests/userland_broker.rs +++ b/litebox_broker_userland/tests/userland_broker.rs @@ -237,7 +237,7 @@ fn run_fake_runner(args: &[OsString]) { assert_eq!(args.len(), 5, "unexpected runner arguments: {args:?}"); let marker = Path::new(&args[4]); let bootstrap = marker.as_os_str().as_encoded_bytes(); - let failed = local.allocate_child_process().unwrap().identity; + let failed = local.allocate_child_process().unwrap(); local .start_child_process( failed.process_id, @@ -246,7 +246,7 @@ fn run_fake_runner(args: &[OsString]) { ) .unwrap(); assert_ne!(failed.process_id.0, failed.initial_thread_id.0); - let started = local.allocate_child_process().unwrap().identity; + let started = local.allocate_child_process().unwrap(); local .start_child_process( started.process_id, diff --git a/litebox_common_linux/src/lib.rs b/litebox_common_linux/src/lib.rs index 7cff5d2ae..687ab6fea 100644 --- a/litebox_common_linux/src/lib.rs +++ b/litebox_common_linux/src/lib.rs @@ -1434,8 +1434,6 @@ pub const TASK_COMM_LEN: usize = 16; pub struct TaskParams { /// Process ID pub pid: i32, - /// Parent Process ID - pub ppid: i32, /// The initial uid. pub uid: u32, /// The initial effective uid. @@ -1979,6 +1977,10 @@ pub enum PrctlArg { SetName(UserPtr), GetName(UserPtrMut), CapBSetRead(usize), + /// PR_SET_CHILD_SUBREAPER: set whether this process adopts orphaned descendants + SetChildSubreaper(usize), + /// PR_GET_CHILD_SUBREAPER: read whether this process adopts orphaned descendants + GetChildSubreaper(UserPtrMut), } #[repr(i32)] @@ -2588,6 +2590,18 @@ pub enum SyscallRequest { }, Getpid, Getppid, + Setpgid { + pid: i32, + pgid: i32, + }, + Getpgid { + pid: i32, + }, + Getpgrp, + Setsid, + Getsid { + pid: i32, + }, Getuid, Geteuid, Getgid, @@ -2965,6 +2979,12 @@ impl SyscallRequest { Sysno::prlimit64 => sys_req!(Prlimit { pid, resource:?, new_limit:*, old_limit:* }), Sysno::getpid => SyscallRequest::Getpid, Sysno::getppid => SyscallRequest::Getppid, + Sysno::setpgid => sys_req!(Setpgid { pid, pgid }), + Sysno::getpgid => sys_req!(Getpgid { pid }), + #[cfg(target_arch = "x86_64")] + Sysno::getpgrp => SyscallRequest::Getpgrp, + Sysno::setsid => SyscallRequest::Setsid, + Sysno::getsid => sys_req!(Getsid { pid }), Sysno::getuid => SyscallRequest::Getuid, Sysno::getgid => SyscallRequest::Getgid, Sysno::geteuid => SyscallRequest::Geteuid, @@ -3024,6 +3044,12 @@ impl SyscallRequest { PrctlOption::CapBSetRead => SyscallRequest::Prctl { args: PrctlArg::CapBSetRead(ctx.sys_req_arg(1)), }, + PrctlOption::SetChildSubreaper => SyscallRequest::Prctl { + args: PrctlArg::SetChildSubreaper(ctx.sys_req_arg(1)), + }, + PrctlOption::GetChildSubreaper => SyscallRequest::Prctl { + args: PrctlArg::GetChildSubreaper(ctx.sys_req_ptr(1)), + }, _ => { return Err(unsupported_einval(format_args!("prctl({op:?})"))); } diff --git a/litebox_common_linux/src/program_startup.rs b/litebox_common_linux/src/program_startup.rs index f5f578931..239a11e58 100644 --- a/litebox_common_linux/src/program_startup.rs +++ b/litebox_common_linux/src/program_startup.rs @@ -20,7 +20,7 @@ use crate::signal::{NSIG, SigAction, SigAltStack, SigSet}; use crate::vmem::VmFlags; use crate::{PtRegs, TASK_COMM_LEN}; -const HEADER_SIZE: usize = size_of::() + size_of::<[u32; 11]>() + size_of::<[u64; 2]>(); +const HEADER_SIZE: usize = size_of::() + size_of::<[u32; 10]>() + size_of::<[u64; 2]>(); /// Size of an inherited descriptor's number, handle, and kind tag, which precede its kind's fields. const INHERITED_FD_HEADER_SIZE: usize = size_of::() + size_of::() + size_of::(); /// Size of a memory region's start, end, and flags. @@ -57,8 +57,6 @@ impl LinuxProcessStartup { /// Platform-managed architectural context is intentionally outside this payload. #[derive(Clone, Debug, PartialEq, Eq)] pub struct LinuxProgramStartup { - /// Parent process ID visible to the child. - pub parent_process_id: i32, /// Real user ID. pub uid: u32, /// Effective user ID. @@ -95,7 +93,7 @@ pub struct InheritedFd { /// Descriptor number. pub fd: u32, /// The child's broker handle to the descriptor's object, as returned by - /// [`litebox::process::Process::inherit`]. + /// [`litebox::process::PendingChild::inherit`]. /// /// Descriptors with the same handle share one open file description, whose kind is taken from /// the first of them. @@ -124,8 +122,6 @@ pub enum InheritedFdKind { /// each region that [has contents](ForkMemoryRegion::has_contents) back to back, in region order. #[derive(Clone)] pub struct LinuxForkStartup { - /// Parent process ID visible to the child. - pub parent_process_id: i32, /// Real user ID. pub uid: u32, /// Effective user ID. @@ -215,9 +211,6 @@ pub enum LinuxProgramStartupError { /// The file mode creation mask has bits other than permission bits. #[error("invalid Linux program umask")] InvalidUmask, - /// The parent process ID is not representable by Linux process semantics. - #[error("invalid Linux program parent process ID")] - InvalidParentProcess, /// An argument or environment string contains an interior NUL. #[error("invalid Linux program string")] InvalidString, @@ -233,7 +226,6 @@ impl LinuxProgramStartup { .try_reserve_exact(encoded_len) .map_err(|_| LinuxProgramStartupError::TooLarge)?; output.push(PROGRAM_STARTUP_TAG); - push_u32(&mut output, self.parent_process_id.cast_unsigned()); push_u32(&mut output, self.uid); push_u32(&mut output, self.euid); push_u32(&mut output, self.gid); @@ -288,7 +280,6 @@ impl LinuxProgramStartup { if read_u8(&mut input)? != PROGRAM_STARTUP_TAG { return Err(LinuxProgramStartupError::Malformed); } - let parent_process_id = read_u32(&mut input)?.cast_signed(); let real_user_id = read_u32(&mut input)?; let effective_user_id = read_u32(&mut input)?; let real_group_id = read_u32(&mut input)?; @@ -347,7 +338,6 @@ impl LinuxProgramStartup { } let envp = values.split_off(argv_count); let startup = Self { - parent_process_id, uid: real_user_id, euid: effective_user_id, gid: real_group_id, @@ -372,7 +362,6 @@ impl LinuxForkStartup { validate_fork(self)?; let mut output = Vec::new(); output.push(FORK_STARTUP_TAG); - push_u32(&mut output, self.parent_process_id.cast_unsigned()); push_u32(&mut output, self.uid); push_u32(&mut output, self.euid); push_u32(&mut output, self.gid); @@ -430,7 +419,6 @@ impl LinuxForkStartup { if read_u8(&mut input)? != FORK_STARTUP_TAG { return Err(LinuxProgramStartupError::Malformed); } - let parent_process_id = read_u32(&mut input)?.cast_signed(); let uid = read_u32(&mut input)?; let euid = read_u32(&mut input)?; let gid = read_u32(&mut input)?; @@ -494,7 +482,6 @@ impl LinuxForkStartup { return Err(LinuxProgramStartupError::Malformed); } let startup = Self { - parent_process_id, uid, euid, gid, @@ -567,9 +554,6 @@ impl InheritedFd { } fn validate(startup: &LinuxProgramStartup) -> Result<(), LinuxProgramStartupError> { - if startup.parent_process_id <= 0 { - return Err(LinuxProgramStartupError::InvalidParentProcess); - } if !startup.path.starts_with('/') || startup.path.as_bytes().contains(&0) { return Err(LinuxProgramStartupError::InvalidPath); } @@ -583,9 +567,6 @@ fn validate(startup: &LinuxProgramStartup) -> Result<(), LinuxProgramStartupErro } fn validate_fork(startup: &LinuxForkStartup) -> Result<(), LinuxProgramStartupError> { - if startup.parent_process_id <= 0 { - return Err(LinuxProgramStartupError::InvalidParentProcess); - } if !startup.cwd.starts_with('/') || startup.cwd.as_bytes().contains(&0) { return Err(LinuxProgramStartupError::InvalidWorkingDirectory); } @@ -723,7 +704,6 @@ mod tests { #[test] fn program_startup_round_trips() { let startup = LinuxProgramStartup { - parent_process_id: 17, uid: 1000, euid: 1001, gid: 1002, @@ -775,7 +755,6 @@ mod tests { #[test] fn program_startup_vector_count_is_payload_bounded() { let startup = LinuxProgramStartup { - parent_process_id: 1, uid: 0, euid: 0, gid: 0, @@ -826,7 +805,6 @@ mod tests { mask: SigSet::empty().with(Signal::SIGUSR2), }; LinuxForkStartup { - parent_process_id: 17, uid: 1000, euid: 1001, gid: 1002, @@ -938,7 +916,6 @@ mod tests { #[test] fn program_startup_rejects_trailing_and_invalid_strings() { let startup = LinuxProgramStartup { - parent_process_id: 1, uid: 0, euid: 0, gid: 0, diff --git a/litebox_runner_linux_on_macos_userland/src/lib.rs b/litebox_runner_linux_on_macos_userland/src/lib.rs index d9cb07839..4ddc27abe 100644 --- a/litebox_runner_linux_on_macos_userland/src/lib.rs +++ b/litebox_runner_linux_on_macos_userland/src/lib.rs @@ -98,7 +98,6 @@ pub fn run(cli_args: CliArgs) -> Result { .load_program( litebox_common_linux::TaskParams { pid: setup.process_id, - ppid: 0, uid: 1000, euid: 1000, gid: 1000, diff --git a/litebox_runner_linux_on_windows_userland/src/lib.rs b/litebox_runner_linux_on_windows_userland/src/lib.rs index 2ed0a3333..1e86b8bcc 100644 --- a/litebox_runner_linux_on_windows_userland/src/lib.rs +++ b/litebox_runner_linux_on_windows_userland/src/lib.rs @@ -114,7 +114,6 @@ pub fn run(cli_args: CliArgs) -> Result<()> { } Some(LinuxProcessStartup::Program(startup)) => { let LinuxProgramStartup { - parent_process_id, uid, euid, gid, @@ -131,7 +130,6 @@ pub fn run(cli_args: CliArgs) -> Result<()> { ( litebox_common_linux::TaskParams { pid: process_id, - ppid: parent_process_id, uid, euid, gid, @@ -179,7 +177,6 @@ pub fn run(cli_args: CliArgs) -> Result<()> { ( litebox_common_linux::TaskParams { pid: process_id, - ppid: 0, uid: 1000, gid: 1000, euid: 1000, diff --git a/litebox_runner_linux_userland/src/lib.rs b/litebox_runner_linux_userland/src/lib.rs index a21989bcc..b10235053 100644 --- a/litebox_runner_linux_userland/src/lib.rs +++ b/litebox_runner_linux_userland/src/lib.rs @@ -183,7 +183,6 @@ fn run_with_seccomp(cli_args: CliArgs, seccomp_scope: SeccompScope) -> Result { let LinuxProgramStartup { - parent_process_id, uid, euid, gid, @@ -200,7 +199,6 @@ fn run_with_seccomp(cli_args: CliArgs, seccomp_scope: SeccompScope) -> Result Result +#include +#include +#include +#include +#include + +static int failures; +static volatile sig_atomic_t usr1_count; +static volatile sig_atomic_t usr2_count; + +#define CHECK(condition) \ + do { \ + if (!(condition)) { \ + printf("failed line=%d errno=%d\n", __LINE__, errno); \ + failures++; \ + } \ + } while (0) + +static void on_usr1(int signal) { + (void)signal; + usr1_count++; +} + +static void on_usr2(int signal) { + (void)signal; + usr2_count++; +} + +// Unblocks `SIGUSR1` and `SIGUSR2` and waits until one of them is handled. +static void wait_for_signal(const sigset_t *blocked) { + sigprocmask(SIG_UNBLOCK, blocked, NULL); + // Sleeping briefly rather than pausing cannot miss a signal handled just before waiting. + const struct timespec delay = {.tv_nsec = 1000000}; + while (usr1_count == 0 && usr2_count == 0) { + nanosleep(&delay, NULL); + } +} + +// Forks a child that joins `process_group`, leads a new group if it is zero, or stays in this +// process's group if it is negative. The child exits with `code` once signaled. +static pid_t fork_member(pid_t process_group, int code, const sigset_t *blocked) { + pid_t child = fork(); + if (child == 0) { + if (process_group >= 0 && setpgid(0, process_group) != 0) { + _exit(1); + } + wait_for_signal(blocked); + _exit(code); + } + // Moving the child from the parent too makes its group known before either continues. + if (child > 0 && process_group >= 0) { + CHECK(setpgid(child, process_group == 0 ? child : process_group) == 0); + } + return child; +} + +// Returns whether `kill(target, 0)` fails with `ESRCH` within five seconds. A reaped child stays +// visible to signals until LiteBox's supervision of it ends shortly after. +static int no_process_remains(pid_t target) { + const struct timespec delay = {.tv_nsec = 1000000}; + for (int i = 0; i < 5000; i++) { + if (kill(target, 0) == -1 && errno == ESRCH) { + return 1; + } + nanosleep(&delay, NULL); + } + return 0; +} + +// Reaps a child selected by `pid`, returning its process ID and storing its exit code. +static pid_t reap(pid_t pid, int *code) { + int status = 0; + pid_t waited; + do { + waited = waitpid(pid, &status, 0); + } while (waited == -1 && errno == EINTR); + *code = WIFEXITED(status) ? WEXITSTATUS(status) : -1; + return waited; +} + +int main(void) { + pid_t self = getpid(); + // LiteBox's initial process leads its own process group and session. + CHECK(getpgrp() == self); + CHECK(getpgid(0) == self); + CHECK(getpgid(self) == self); + CHECK(getsid(0) == self); + CHECK(setsid() == -1 && errno == EPERM); + CHECK(getpgid(-1) == -1 && errno == ESRCH); + CHECK(getsid(-1) == -1 && errno == ESRCH); + CHECK(getpgid(0x7fffff) == -1 && errno == ESRCH); + CHECK(setpgid(0, -1) == -1 && errno == EINVAL); + // Like Linux, a session leader cannot change its process group. + CHECK(setpgid(0, 0) == -1 && errno == EPERM); + + struct sigaction action = {0}; + action.sa_handler = on_usr1; + sigaction(SIGUSR1, &action, NULL); + action.sa_handler = on_usr2; + sigaction(SIGUSR2, &action, NULL); + // Children start with both signals blocked, so none arrives before they wait for it. + sigset_t blocked; + sigemptyset(&blocked); + sigaddset(&blocked, SIGUSR1); + sigaddset(&blocked, SIGUSR2); + sigprocmask(SIG_BLOCK, &blocked, NULL); + + // Two children in a new group and one in this process's group. + pid_t leader = fork_member(0, 10, &blocked); + pid_t member = fork_member(leader, 20, &blocked); + pid_t sibling = fork_member(-1, 30, &blocked); + CHECK(leader > 0 && member > 0 && sibling > 0); + CHECK(getpgid(leader) == leader); + CHECK(getpgid(member) == leader); + CHECK(getpgid(sibling) == self); + CHECK(getsid(member) == self); + // A group must exist in the caller's session to be joined. + CHECK(setpgid(sibling, 0x7fffff) == -1 && errno == EPERM); + CHECK(kill(-leader, 0) == 0); + CHECK(kill(-1, 0) == 0); + + // Signaling this process's group delivers to this process before `kill` returns. + sigprocmask(SIG_UNBLOCK, &blocked, NULL); + CHECK(kill(0, SIGUSR2) == 0); + CHECK(usr2_count == 1); + CHECK(usr1_count == 0); + + int code = 0; + // Waiting for this process's group skips the other group's children. + CHECK(reap(0, &code) == sibling && code == 30); + CHECK(waitpid(0, NULL, WNOHANG) == -1 && errno == ECHILD); + + CHECK(kill(-leader, SIGUSR1) == 0); + CHECK(usr1_count == 0); + int codes = 0; + for (int i = 0; i < 2; i++) { + pid_t waited = reap(-leader, &code); + CHECK(waited == leader || waited == member); + codes += code; + } + CHECK(codes == 30); + CHECK(waitpid(-leader, NULL, 0) == -1 && errno == ECHILD); + CHECK(no_process_remains(-leader)); + + // A forked child starts a session, after which it can neither start another nor be moved. + int ready[2]; + int release[2]; + CHECK(pipe(ready) == 0 && pipe(release) == 0); + pid_t session = fork(); + if (session == 0) { + close(ready[0]); + close(release[1]); + char byte = 0; + int ok = setsid() == getpid() && getsid(0) == getpid() && getpgrp() == getpid(); + ok = ok && setsid() == -1 && errno == EPERM; + ok = ok && setpgid(0, self) == -1 && errno == EPERM; + ok = ok && write(ready[1], &byte, 1) == 1 && read(release[0], &byte, 1) == 0; + _exit(ok ? 40 : 41); + } + close(ready[1]); + close(release[0]); + char byte; + CHECK(read(ready[0], &byte, 1) == 1); + CHECK(getsid(session) == session); + CHECK(getpgid(session) == session); + CHECK(setpgid(session, session) == -1 && errno == EPERM); + close(release[1]); + CHECK(reap(session, &code) == session && code == 40); + + // A vforked child starts a session before it exits. + pid_t vforked = vfork(); + if (vforked == 0) { + _exit(setsid() == getpid() && getsid(0) == getpid() && getpgid(0) == getpid() ? 50 : 51); + } + CHECK(reap(vforked, &code) == vforked && code == 50); + + // No other process remains to signal. + CHECK(no_process_remains(-1)); + + printf("process-group failures=%d\n", failures); + return failures == 0 ? 0 : 1; +} diff --git a/litebox_runner_linux_userland/tests/process_tree_parent.c b/litebox_runner_linux_userland/tests/process_tree_parent.c new file mode 100644 index 000000000..73761c54d --- /dev/null +++ b/litebox_runner_linux_userland/tests/process_tree_parent.c @@ -0,0 +1,169 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +// Checks parents, orphan adoption, and child subreapers across forked children. + +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include + +static int failures; +static volatile sig_atomic_t sigchld_count; + +#define CHECK(condition) \ + do { \ + if (!(condition)) { \ + printf("failed line=%d errno=%d\n", __LINE__, errno); \ + failures++; \ + } \ + } while (0) + +static void on_sigchld(int signal) { + (void)signal; + sigchld_count++; +} + +// Reaps a child selected by `pid`, returning its process ID and storing its exit code. +static pid_t reap(pid_t pid, int *code) { + int status = 0; + pid_t waited; + do { + waited = waitpid(pid, &status, 0); + } while (waited == -1 && errno == EINTR); + *code = WIFEXITED(status) ? WEXITSTATUS(status) : -1; + return waited; +} + +// Reads a process ID from `report`, waiting up to five seconds for it, so a child that fails to +// report one fails the test instead of hanging it. +static int read_report(int report, pid_t *pid) { + struct pollfd readable = {.fd = report, .events = POLLIN}; + int ready; + do { + ready = poll(&readable, 1, 5000); + } while (ready == -1 && errno == EINTR); + return ready == 1 && read(report, pid, sizeof *pid) == sizeof *pid; +} + +// Waits up to five seconds for this process's parent to stop being `parent`, returning the new +// parent. +static pid_t wait_for_new_parent(pid_t parent) { + const struct timespec delay = {.tv_nsec = 1000000}; + for (int i = 0; i < 5000 && getppid() == parent; i++) { + nanosleep(&delay, NULL); + } + return getppid(); +} + +// Returns the subreaper attribute of this process, or -1 if it cannot be read. +static int subreaper(void) { + int value = -1; + return prctl(PR_GET_CHILD_SUBREAPER, &value) == 0 ? value : -1; +} + +// Forks a child that forks an orphan and exits, writing the orphan's ID to `report`. The orphan +// exits with `adopted_code` if `adopter` adopts it, and the child exits with `parent_code`. +static pid_t orphan_child(pid_t adopter, int report, int parent_code, int adopted_code) { + pid_t parent = fork(); + if (parent == 0) { + pid_t self = getpid(); + // Like Linux, a child does not inherit its parent's subreaper attribute. + int code = subreaper() == 0 ? parent_code : 1; + pid_t orphan = fork(); + if (orphan == 0) { + _exit(wait_for_new_parent(self) == adopter ? adopted_code : 2); + } + if (orphan < 0 || write(report, &orphan, sizeof orphan) != sizeof orphan) { + code = 3; + } + _exit(code); + } + return parent; +} + +// Forks a child that forks a grandchild and exits only once the grandchild has exited, so the +// grandchild is orphaned as a zombie. Writes the grandchild's ID to `report`. +static pid_t zombie_child(int report, int parent_code, int zombie_code) { + pid_t parent = fork(); + if (parent == 0) { + struct sigaction action = {0}; + action.sa_handler = on_sigchld; + sigaction(SIGCHLD, &action, NULL); + int code = parent_code; + pid_t zombie = fork(); + if (zombie == 0) { + _exit(zombie_code); + } + if (zombie < 0 || write(report, &zombie, sizeof zombie) != sizeof zombie) { + code = 3; + } + // The grandchild has terminated once its `SIGCHLD` is handled. + const struct timespec delay = {.tv_nsec = 1000000}; + for (int i = 0; i < 5000 && sigchld_count == 0; i++) { + nanosleep(&delay, NULL); + } + _exit(sigchld_count == 0 ? 4 : code); + } + return parent; +} + +int main(void) { + pid_t self = getpid(); + int code = 0; + int report[2]; + CHECK(pipe(report) == 0); + + // LiteBox's initial process has no parent, like a container's init. + CHECK(getppid() == 0); + CHECK(subreaper() == 0); + pid_t child = fork(); + if (child == 0) { + _exit(getppid() == self ? 10 : 11); + } + CHECK(reap(child, &code) == child && code == 10); + + // The initial process adopts orphans even after setting and clearing the subreaper attribute. + CHECK(prctl(PR_SET_CHILD_SUBREAPER, 1) == 0 && subreaper() == 1); + CHECK(prctl(PR_SET_CHILD_SUBREAPER, 0) == 0 && subreaper() == 0); + pid_t parent = orphan_child(self, report[1], 20, 21); + pid_t orphan = -1; + CHECK(read_report(report[0], &orphan)); + CHECK(reap(parent, &code) == parent && code == 20); + CHECK(reap(orphan, &code) == orphan && code == 21); + + // An orphaned zombie is reported to its adopter. + parent = zombie_child(report[1], 30, 31); + pid_t zombie = -1; + CHECK(read_report(report[0], &zombie)); + CHECK(reap(parent, &code) == parent && code == 30); + CHECK(reap(zombie, &code) == zombie && code == 31); + + // A subreaper adopts the orphans of its descendants instead of the initial process. + pid_t reaper = fork(); + if (reaper == 0) { + pid_t reaper_self = getpid(); + int ok = getppid() == self && subreaper() == 0; + ok = ok && prctl(PR_SET_CHILD_SUBREAPER, 1) == 0 && subreaper() == 1; + pid_t reaper_parent = orphan_child(reaper_self, report[1], 40, 41); + pid_t reaper_orphan = -1; + int reaper_code = 0; + ok = ok && read_report(report[0], &reaper_orphan); + ok = ok && reap(reaper_parent, &reaper_code) == reaper_parent && reaper_code == 40; + ok = ok && reap(reaper_orphan, &reaper_code) == reaper_orphan && reaper_code == 41; + ok = ok && prctl(PR_SET_CHILD_SUBREAPER, 0) == 0 && subreaper() == 0; + _exit(ok ? 50 : 51); + } + CHECK(reap(reaper, &code) == reaper && code == 50); + + // No child remains to wait for. + CHECK(waitpid(-1, NULL, WNOHANG) == -1 && errno == ECHILD); + + printf("process-tree failures=%d\n", failures); + return failures == 0 ? 0 : 1; +} diff --git a/litebox_runner_linux_userland/tests/run.rs b/litebox_runner_linux_userland/tests/run.rs index cd1515f65..aac1b379c 100644 --- a/litebox_runner_linux_userland/tests/run.rs +++ b/litebox_runner_linux_userland/tests/run.rs @@ -65,6 +65,8 @@ const DEDICATED_C_TESTS: &[&str] = &[ "fork_threads_parent.c", "fork_aarch64_state.c", "gate_signals.c", + "process_group_parent.c", + "process_tree_parent.c", "sigreturn.c", "sigreturn_simd.c", "svc_scratch_regs.c", @@ -1068,6 +1070,46 @@ fn fork_requires_process_duplication() { assert_eq!(numeric_field(line, "errno="), libc::EPERM); } +#[cfg(target_os = "linux")] +#[test] +fn process_groups_and_sessions_span_processes() { + let parent = common::compile( + "./tests/process_group_parent.c", + "process_group_parent", + true, + false, + ); + let mut runner = Runner::new(&parent, "process_group_parent"); + runner.allow_process_duplication(); + + let output = String::from_utf8(runner.output()).unwrap(); + assert!( + output + .lines() + .any(|line| line == "process-group failures=0"), + "{output}" + ); +} + +#[cfg(target_os = "linux")] +#[test] +fn orphans_are_adopted_by_subreapers_and_the_initial_process() { + let parent = common::compile( + "./tests/process_tree_parent.c", + "process_tree_parent", + true, + false, + ); + let mut runner = Runner::new(&parent, "process_tree_parent"); + runner.allow_process_duplication(); + + let output = String::from_utf8(runner.output()).unwrap(); + assert!( + output.lines().any(|line| line == "process-tree failures=0"), + "{output}" + ); +} + #[cfg(target_os = "linux")] #[test] fn forked_shell_subshells_and_pipelines() { diff --git a/litebox_shim_linux/src/lib.rs b/litebox_shim_linux/src/lib.rs index 411849d8b..4f0931921 100644 --- a/litebox_shim_linux/src/lib.rs +++ b/litebox_shim_linux/src/lib.rs @@ -433,7 +433,6 @@ impl LinuxShim { ) -> Result, loader::elf::ElfLoaderError> { let litebox_common_linux::TaskParams { pid, - ppid, uid, euid, gid, @@ -444,7 +443,7 @@ impl LinuxShim { cwd, umask, } = task; - if pid != self.0.process_id || ppid < 0 { + if pid != self.0.process_id { return Err(loader::elf::ElfLoaderError::InvalidProcessId); } @@ -486,7 +485,6 @@ impl LinuxShim { wait_state: wait::WaitState::new(self.0.platform), vfork: RefCell::new(None), pid, - ppid, credentials, comm: [0; litebox_common_linux::TASK_COMM_LEN].into(), // set at load time fs: fs_state.into(), @@ -499,6 +497,7 @@ impl LinuxShim { if entrypoints.task.signals.reaps_children() { entrypoints.task.set_child_reaping(true); } + entrypoints.task.adopt_orphans_if_first(); entrypoints.task.open_signals(); let (path, argv) = entrypoints @@ -534,7 +533,6 @@ impl LinuxShim { load_image: impl FnMut(u64, &mut [u8]) -> Result<(), Errno>, ) -> Result, ForkRestoreError> { let litebox_common_linux::program_startup::LinuxForkStartup { - parent_process_id, uid, euid, gid, @@ -600,7 +598,6 @@ impl LinuxShim { wait_state: wait::WaitState::new(self.0.platform), vfork: RefCell::new(None), pid, - ppid: parent_process_id, credentials, comm: comm.into(), fs: Arc::new(fs_state).into(), @@ -863,8 +860,9 @@ impl Task { let syscall_number = ctx.syscallno.cast_unsigned() as usize; let request = SyscallRequest::try_from_raw(syscall_number, ctx, log_unsupported_fmt); // The constrained vfork child may only inspect its temporary identity and resource limits, - // manage its own signal state, descriptors, working directory, and umask, open and write - // files, exit, or attempt execve. Any other syscall terminates the shared runner. + // manage its own process group and session, signal state, descriptors, working directory, + // and umask, open and write files, exit, or attempt execve. Any other syscall terminates + // the shared runner. let is_vfork_child = self.vfork.borrow().is_some(); if is_vfork_child && !matches!( @@ -875,6 +873,11 @@ impl Task { | SyscallRequest::Getpid | SyscallRequest::Getppid | SyscallRequest::Gettid + | SyscallRequest::Setpgid { .. } + | SyscallRequest::Getpgid { .. } + | SyscallRequest::Getpgrp + | SyscallRequest::Setsid + | SyscallRequest::Getsid { .. } | SyscallRequest::Prlimit { pid: 0, new_limit: None, @@ -1445,6 +1448,11 @@ impl Task { } => self.sys_wait4(pid, wstatus, options, rusage), SyscallRequest::Getpid => Ok(self.sys_getpid().reinterpret_as_unsigned() as usize), SyscallRequest::Getppid => Ok(self.sys_getppid().reinterpret_as_unsigned() as usize), + SyscallRequest::Setpgid { pid, pgid } => self.sys_setpgid(pid, pgid), + SyscallRequest::Getpgid { pid } => self.sys_getpgid(pid), + SyscallRequest::Getpgrp => self.sys_getpgrp(), + SyscallRequest::Setsid => self.sys_setsid(), + SyscallRequest::Getsid { pid } => self.sys_getsid(pid), SyscallRequest::Getuid => Ok(self.sys_getuid() as usize), SyscallRequest::Getgid => Ok(self.sys_getgid() as usize), SyscallRequest::Geteuid => Ok(self.sys_geteuid() as usize), @@ -1537,8 +1545,6 @@ struct Task { vfork: RefCell>>, /// Process ID pid: i32, - /// Parent Process ID - ppid: i32, /// Task credentials. These are set per task but are Arc'd to save space /// since most tasks never change their credentials. credentials: Arc, @@ -1553,7 +1559,7 @@ struct Task { } struct VforkState { - child: litebox::process::Process, + child: litebox::process::PendingChild, child_pid: i32, parent_context: litebox_common_linux::PtRegs, /// The parent's floating-point and vector state, which the child may change. @@ -1618,7 +1624,6 @@ mod test_utils { thread: syscalls::process::ThreadState::new_process(pid), vfork: RefCell::new(None), pid, - ppid: 0, credentials, comm: Cell::new(*b"test\0\0\0\0\0\0\0\0\0\0\0\0"), fs: fs_state.into(), @@ -1646,7 +1651,6 @@ mod test_utils { thread, vfork: RefCell::new(None), pid: self.pid, - ppid: self.ppid, credentials: self.credentials.clone(), comm: self.comm.clone(), fs: self.fs.clone(), diff --git a/litebox_shim_linux/src/syscalls/process.rs b/litebox_shim_linux/src/syscalls/process.rs index f550e81ec..728869700 100644 --- a/litebox_shim_linux/src/syscalls/process.rs +++ b/litebox_shim_linux/src/syscalls/process.rs @@ -22,11 +22,14 @@ use litebox::event::{Events, IOPollable as _}; use litebox::platform::ArchSpecificRegister; use litebox::platform::RawConstPointer as _; use litebox::platform::TimerHandle; -use litebox::process::{ChildStatus, ProcessError}; +use litebox::process::ProcessError; use litebox::sync::{Mutex, RwLock}; use litebox::utils::TruncateExt as _; +use litebox_broker_protocol::ProcessId; use litebox_broker_protocol::process::MAX_CHILD_MEMORY_WRITE_SIZE; -use litebox_broker_protocol::process::ProcessExitStatus; +use litebox_broker_protocol::process::{ChildSelector, ProcessExitStatus}; +use litebox_broker_protocol::process_group::ProcessGroupMembership; +use litebox_broker_protocol::signal::SignalEvent; use litebox_common_linux::ProtFlags; use litebox_common_linux::program_startup::{ForkMemoryRegion, LinuxForkStartup}; use litebox_common_linux::signal::{CLD_EXITED, Signal}; @@ -222,11 +225,13 @@ pub(crate) struct ProcessState { /// the `inner` mutex lock. fork_pause: ::RawMutex, inner: Arc>>, - /// Started child processes that have not been reaped, mapped by process ID. - children: Mutex>>, - /// Termination events of the children, shared with each child's watcher. - child_events: Arc>, - /// Signals other processes send to this process, once opened. + /// Notified when a child's exit or removal is taken from `signals`, waking the threads waiting + /// for a child. + child_exits: Pollee, + /// Whether this process is a child subreaper (`PR_SET_CHILD_SUBREAPER`). Held while the + /// broker is updated, so concurrent changes apply in order. + child_subreaper: Mutex, + /// Signals other processes send to this process and its children's exits, once opened. signals: once_cell::race::OnceBox>, /// Watcher of `signals`. signal_watcher: Arc>, @@ -238,57 +243,14 @@ pub(crate) struct ProcessState { pub(crate) sigreturn_trampoline: Mutex>, } -/// A started child process. -struct Child { - process: litebox::process::Process, - /// Watcher of the child's termination, registered on `process`. - watcher: Arc>, - /// Termination status once observed. Observing a termination queues the child's `SIGCHLD`. - exit_status: Option, -} - -/// A process's view of its children's terminations, which its threads observe. -/// -/// This does not own the children, so a watcher holding it cannot drop them while its child's -/// handle notifies it. -struct ChildEvents { - /// Set when a child may have terminated since child terminations were last observed. - changed: AtomicBool, - /// Notified when a child may have terminated, waking the threads waiting for a child. - pollee: Pollee, - /// The process's locked state, to interrupt its threads. - process: Arc>>, -} - -/// Watcher of one child's termination, which wakes and interrupts its parent's threads to observe -/// it. -struct ChildWatcher { - /// Set when the child may have terminated since its status was last queried. - changed: AtomicBool, - /// The parent's view of its children's terminations. - parent: Arc>, -} - -impl Observer for ChildWatcher { - fn on_events(&self, events: &Events) { - // Closing a child's handle wakes its observers without events, and the closing thread - // already removed that child. - if events.is_empty() { - return; - } - // Set the child's flag first, so an observation that sees the parent's also sees it. - self.changed.store(true, Ordering::SeqCst); - self.parent.changed.store(true, Ordering::SeqCst); - self.parent.pollee.notify_observers(Events::IN); - interrupt_threads(&self.parent.process, None); - } -} - -/// Watcher of the signals other processes send to a process, which interrupts its threads to take -/// them. +/// Watcher of the signals other processes send to a process and its children's exits, which +/// interrupts its threads to take them. struct SignalWatcher { - /// Set when a signal may have been sent since signals were last taken. + /// Set when a signal or child exit may have arrived since signals were last taken. changed: AtomicBool, + /// Held while taking signals, so a forced take waits for one in progress to queue the + /// signals it took. + taking: Mutex, /// The process's locked state, to interrupt its threads. process: Arc>>, } @@ -379,19 +341,16 @@ impl ProcessState { Self { nr_threads, fork_pause: ::RawMutex::INIT, - child_events: Arc::new(ChildEvents { - changed: AtomicBool::new(false), - pollee: Pollee::new(), - process: inner.clone(), - }), + child_exits: Pollee::new(), + child_subreaper: Mutex::new(false), signals: once_cell::race::OnceBox::new(), // Signals sent before they are opened are taken at the first check. signal_watcher: Arc::new(SignalWatcher { changed: AtomicBool::new(true), + taking: Mutex::new(()), process: inner.clone(), }), inner, - children: Mutex::new(BTreeMap::new()), limits: ResourceLimits::default(), alarm_timer: Mutex::new(Alarm { handle: None, @@ -412,103 +371,6 @@ impl ProcessState { alarm.handle.is_none() && alarm.deadline.is_none() } - /// Adds a started child process and watches its termination. - /// - /// The child may have terminated before its watcher was registered, so the watcher starts - /// changed and the caller then observes child terminations. - fn add_child(&self, pid: i32, process: litebox::process::Process) { - let watcher = Arc::new(ChildWatcher { - changed: AtomicBool::new(true), - parent: self.child_events.clone(), - }); - process.register_observer( - Arc::downgrade(&watcher) as Weak>, - Events::IN, - ); - let previous = self.children.lock().insert( - pid, - Child { - process, - watcher, - exit_status: None, - }, - ); - assert!( - previous.is_none(), - "broker child process IDs must be unique" - ); - } - - /// Records the termination of each child whose termination was not yet observed and whose - /// watcher saw a change since its status was last queried, returning their process IDs and - /// statuses. - /// - /// Children the broker reaped as they terminated are released, since no wait reports them - /// and each holds broker process capacity until its handle closes. Nothing is recorded if a - /// status query fails. - fn record_child_terminations(&self) -> Result, ProcessError> { - // Clear each flag before querying, so a termination published during the queries sets it - // again. - self.child_events.changed.store(false, Ordering::SeqCst); - let mut children = self.children.lock(); - let statuses = children - .iter() - .filter(|(_, child)| { - child.exit_status.is_none() && child.watcher.changed.swap(false, Ordering::SeqCst) - }) - .map(|(&pid, child)| Ok((pid, child.process.status()?))) - .collect::, ProcessError>>()?; - let mut terminated = Vec::new(); - let mut reaped = Vec::new(); - for (pid, status) in statuses { - match status { - ChildStatus::Live => {} - ChildStatus::Terminated(exit_status) => { - children - .get_mut(&pid) - .expect("observed child must remain present") - .exit_status = Some(exit_status); - terminated.push((pid, exit_status)); - } - ChildStatus::Reaped(exit_status) => { - reaped.push(children.remove(&pid)); - terminated.push((pid, exit_status)); - } - } - } - drop(children); - drop(reaped); - Ok(terminated) - } - - /// Removes one child with process ID `target`, or any child when `target` is `None`, whose - /// termination was observed. - /// - /// The caller drops the returned child outside the children lock to reap it. - fn take_exited_child( - &self, - target: Option, - ) -> Result<(i32, litebox::process::Process, ProcessExitStatus), TryOpError> - { - let mut children = self.children.lock(); - let mut matching = children - .iter() - .filter(|&(&pid, _)| target.is_none_or(|target| target == pid)) - .peekable(); - if matching.peek().is_none() { - return Err(TryOpError::Other(Errno::ECHILD)); - } - let Some((pid, exit_status)) = - matching.find_map(|(&pid, child)| child.exit_status.map(|status| (pid, status))) - else { - return Err(TryOpError::TryAgain); - }; - let child = children - .remove(&pid) - .expect("matched child must remain present"); - Ok((pid, child.process, exit_status)) - } - /// Waits for all threads in this process to exit, returning the exit code. pub fn wait_for_exit(&self) -> ExitStatus { loop { @@ -609,45 +471,8 @@ impl Task { interrupt_threads(&self.thread.process.inner, Some(self.tid())); } - /// Queues `SIGCHLD` for each child whose termination was not yet observed. - /// - /// During a `vfork` window the signal state is the child's, so the parent observes the - /// terminations once it resumes. - pub(crate) fn observe_child_terminations(&self) -> Result<(), ProcessError> { - if self.vfork.borrow().is_some() { - return Ok(()); - } - let terminated = self.thread.process.record_child_terminations()?; - // Like Linux, a child terminating while `SIGCHLD` is set to `SIG_IGN` sends no signal, - // even if `SIGCHLD` is blocked. - if terminated.is_empty() || self.signals.ignored().contains(Signal::SIGCHLD) { - return Ok(()); - } - for (pid, exit_status) in terminated { - self.send_shared_signal( - Signal::SIGCHLD, - siginfo_child(pid, self.credentials.uid, exit_status), - ); - } - Ok(()) - } - - /// Queues `SIGCHLD` for children that may have terminated since child terminations were last - /// observed. - pub(crate) fn check_for_child_terminations(&self) { - if self - .thread - .process - .child_events - .changed - .load(Ordering::SeqCst) - { - // Failure means the process service failed, so no termination can be observed. - let _ = self.observe_child_terminations(); - } - } - - /// Opens the signals other processes send to this process, so its threads take them. + /// Opens the signals other processes send to this process and its children's exits, so its + /// threads take them. pub(crate) fn open_signals(&self) { let process = &self.thread.process; // Failure means the process service is unavailable or has failed, so no signal can be @@ -665,11 +490,16 @@ impl Task { ); } - /// Queues the signals other processes sent since signals were last taken. + /// Queues the signals other processes sent, and `SIGCHLD` for the children that exited, since + /// signals were last taken, waking the threads waiting for a child. + /// + /// Unless `force` is set, this only takes signals after being notified of them. A forced take + /// does not wait for the notification, so, like Linux, a process that signals its own process + /// group handles the signal before `kill` returns. /// /// During a `vfork` window the signal state is the child's, so the parent takes the signals /// once it resumes. - pub(crate) fn take_signals(&self) { + pub(crate) fn take_signals(&self, force: bool) { let process = &self.thread.process; if self.vfork.borrow().is_some() { return; @@ -677,24 +507,70 @@ impl Task { let Some(signals) = process.signals.get() else { return; }; - if !process.signal_watcher.changed.swap(false, Ordering::SeqCst) { + let watcher = &process.signal_watcher; + if !force && !watcher.changed.load(Ordering::SeqCst) { + return; + } + // A forced take waits here for a take in progress to queue the signals it took. + let _taking = watcher.taking.lock(); + if !watcher.changed.swap(false, Ordering::SeqCst) && !force { return; } // Failure means the process service failed, so no more signals can be taken. - while let Ok(Some(received)) = signals.take() { - if let Ok(signal) = Signal::try_from(received.signal.cast_signed()) { - self.send_shared_signal( - signal, - siginfo_kill_from( - signal, - received.sender.0.cast_signed(), - self.credentials.uid, - ), - ); + while let Ok(Some(event)) = signals.take() { + match event { + SignalEvent::Signal(received) => { + if let Ok(signal) = Signal::try_from(received.signal.cast_signed()) { + self.send_shared_signal( + signal, + siginfo_kill_from( + signal, + received.sender.0.cast_signed(), + self.credentials.uid, + ), + ); + } + } + SignalEvent::ChildExited(exit) => { + // Like Linux, a child terminating while `SIGCHLD` is set to `SIG_IGN` sends + // no signal, even if `SIGCHLD` is blocked. The broker coalesces exits, so one + // `SIGCHLD` may stand for several, as when a pending `SIGCHLD` absorbs others. + if !self.signals.ignored().contains(Signal::SIGCHLD) { + self.send_shared_signal( + Signal::SIGCHLD, + siginfo_child( + exit.process_id.0.cast_signed(), + self.credentials.uid, + exit.exit_status, + ), + ); + } + process.child_exits.notify_observers(Events::IN); + } + // A child that never started sends no `SIGCHLD`, but waits for it must end. + SignalEvent::ChildRemoved => process.child_exits.notify_observers(Events::IN), } } } + /// Makes this process adopt the orphans of its descendants if it is the first process, which, + /// like Linux init, no process created. + pub(crate) fn adopt_orphans_if_first(&self) { + // Failure means the process service is unavailable or has failed, so no child can be + // created. + if self.is_first_process() { + let _ = self.global.litebox.set_orphan_adoption(true); + } + } + + /// Returns whether this process is the first process, which no process created. + fn is_first_process(&self) -> bool { + self.global + .litebox + .process_info(ProcessId(self.pid.cast_unsigned())) + .is_ok_and(|info| info.creator.is_none()) + } + /// Updates the process exit status for a thread exit. fn exit_thread(&self, code: i8) { let mut inner = self.thread.process.inner.lock(); @@ -929,6 +805,28 @@ impl Task { // Note we don't support capabilities in LiteBox, so we always return 0. Ok(0) } + PrctlArg::SetChildSubreaper(enabled) => { + let enabled = enabled != 0; + let mut subreaper = self.thread.process.child_subreaper.lock(); + // The first process adopts orphans regardless, like Linux init. + match self + .global + .litebox + .set_orphan_adoption(enabled || self.is_first_process()) + { + Ok(()) | Err(ProcessError::Unavailable) => {} + Err(error) => return Err(error.into()), + } + *subreaper = enabled; + Ok(0) + } + PrctlArg::GetChildSubreaper(subreaper) => { + let enabled = *self.thread.process.child_subreaper.lock(); + subreaper + .write_at_offset::(0, i32::from(enabled)) + .ok_or(Errno::EFAULT) + .map(|()| 0) + } _ => unimplemented!(), } } @@ -1094,9 +992,6 @@ impl Task { self.files.replace(state.parent_files); self.fs.replace(state.parent_fs); self.signals.restore_vfork_parent(state.parent_signals); - self.thread.process.add_child(state.child_pid, state.child); - // Failure means the process service failed, so no termination can be observed. - let _ = self.observe_child_terminations(); *ctx = state.parent_context; self.global .platform @@ -1256,7 +1151,6 @@ impl Task { let (blocked_signals, signal_actions, alternate_signal_stack) = self.signals.fork_state(); let (initial_program_break, program_break) = self.global.mm.program_break(); let mut startup = LinuxForkStartup { - parent_process_id: self.pid, uid: self.credentials.uid, euid: self.credentials.euid, gid: self.credentials.gid, @@ -1284,8 +1178,7 @@ impl Task { // The child keeps the references it inherits until it exits, so check the size first. startup.encode().map_err(|_| Errno::ENOMEM)?; - // Release the children the broker reaped, which hold process capacity until observed. - let _ = self.observe_child_terminations(); + // Dropping the pending child on failure discards it. let child = self .global .litebox @@ -1293,21 +1186,10 @@ impl Task { .map_err(Errno::from)?; let child_pid = i32::try_from(child.identity().process_id.0) .expect("broker process IDs must fit Linux pid_t"); - let prepared = self - .write_fork_image(&child, &startup.regions) - .and_then(|()| { - child - .inherit(&self.global.litebox, &objects) - .map_err(Errno::from) - }); - let handles = match prepared { - Ok(handles) => handles, - Err(errno) => { - // The pending child never runs, and no one observes its status. - let _ = child.exit(ProcessExitStatus::Unknown); - return Err(errno); - } - }; + self.write_fork_image(&child, &startup.regions)?; + let handles = child + .inherit(&self.global.litebox, &objects) + .map_err(Errno::from)?; for (forked, handle) in startup.fds.iter_mut().zip(handles) { forked.inherited.handle = handle; } @@ -1316,11 +1198,7 @@ impl Task { let payload = startup .encode() .expect("the startup fit with placeholder handles"); - // The broker no longer holds a pending child if starting it fails. child.start(&payload).map_err(Errno::from)?; - self.thread.process.add_child(child_pid, child); - // Failure means the process service failed, so no termination can be observed. - let _ = self.observe_child_terminations(); Ok(child_pid.cast_unsigned() as usize) } @@ -1328,7 +1206,7 @@ impl Task { /// process image, back to back, skipping zero-filled pages at the ends of each chunk. fn write_fork_image( &self, - child: &litebox::process::Process, + child: &litebox::process::PendingChild, regions: &[ForkMemoryRegion], ) -> Result<(), Errno> { const CHUNK_SIZE: usize = MAX_CHILD_MEMORY_WRITE_SIZE as usize; @@ -1471,8 +1349,6 @@ impl Task { { return Err(Errno::EAGAIN); } - // Release the children the broker reaped, which hold process capacity until observed. - let _ = self.observe_child_terminations(); let child = self .global .litebox @@ -1735,7 +1611,6 @@ impl Task { thread, vfork: core::cell::RefCell::new(None), pid: self.pid, - ppid: self.ppid, credentials: self.credentials.clone(), comm: self.comm.clone(), fs: fs.into(), @@ -2277,8 +2152,8 @@ impl Task { /// Handle syscall `wait4`. /// /// Only terminated children are reported. Like Linux, children that terminate while - /// `SIGCHLD` is ignored or has `SA_NOCLDWAIT` are reaped instead of reported. Process groups - /// are not modeled, so `pid == 0` waits for any child and `pid < -1` matches no child. + /// `SIGCHLD` is ignored or has `SA_NOCLDWAIT` are reaped instead of reported, and `pid == 0` + /// waits for the children in the process group this process is in when it starts waiting. /// Resource usage is reported as zero. /// A child without an observable termination status is reported as killed by `SIGSEGV`. /// Children belong to the process rather than the creating thread, so `__WNOTHREAD` is @@ -2300,11 +2175,17 @@ impl Task { if options & !(WNOHANG | WUNTRACED | WCONTINUED | WNOTHREAD | WALL | WCLONE) != 0 { return Err(Errno::EINVAL); } - let target = match pid { - -1 | 0 => None, - 1.. => Some(pid), + let selector = match pid { + -1 => ChildSelector::Any, + 1.. => ChildSelector::Process(ProcessId(pid.cast_unsigned())), i32::MIN => return Err(Errno::ESRCH), - _ => return Err(Errno::ECHILD), + 0 => match self.membership(self.pid) { + Ok(membership) => ChildSelector::ProcessGroup(membership.process_group), + // Without a process service, this process has no children. + Err(ProcessError::Unavailable) => return Err(Errno::ECHILD), + Err(error) => return Err(error.into()), + }, + _ => ChildSelector::ProcessGroup(ProcessId(pid.unsigned_abs())), }; // Every child is created with the default exit signal, so only __WALL selects it // together with __WCLONE. @@ -2312,33 +2193,35 @@ impl Task { return Err(Errno::ECHILD); } let process = &self.thread.process; - let mut take_exited_child = || { - self.observe_child_terminations() - .map_err(|error| TryOpError::Other(error.into()))?; - process.take_exited_child(target) + let mut reap_child = || match self.global.litebox.reap_child(selector) { + Ok(Some(exit)) => Ok(exit), + Ok(None) => Err(TryOpError::TryAgain), + // Without a process service, or a matching child, there is nothing to wait for. + Err(ProcessError::Unavailable | ProcessError::NoSuchProcess) => { + Err(TryOpError::Other(Errno::ECHILD)) + } + Err(error) => Err(TryOpError::Other(error.into())), }; - let exited = match process.child_events.pollee.wait( + let exited = match process.child_exits.wait( &self.wait_cx(), options & WNOHANG != 0, Events::IN, - &mut take_exited_child, + &mut reap_child, ) { // Like Linux, which checks for terminated children before pending signals, report a // child whose termination queued the interrupting `SIGCHLD`. - Err(TryOpError::WaitError(WaitError::Interrupted)) => match take_exited_child() { + Err(TryOpError::WaitError(WaitError::Interrupted)) => match reap_child() { Err(TryOpError::TryAgain) => Err(TryOpError::WaitError(WaitError::Interrupted)), result => result, }, result => result, }; - let (child_pid, child, exit_status) = match exited { - Ok(exited) => exited, + let exit = match exited { + Ok(exit) => exit, Err(TryOpError::TryAgain) => return Ok(0), Err(error) => return Err(error.into()), }; - // Closing the child handle reaps the child. - drop(child); - let (code, status) = child_termination(exit_status); + let (code, status) = child_termination(exit.exit_status); let status = if code == CLD_EXITED { status << 8 } else { @@ -2354,7 +2237,77 @@ impl Task { .write_at_offset::(0, Rusage::default()) .ok_or(Errno::EFAULT)?; } - Ok(child_pid.cast_unsigned() as usize) + Ok(exit.process_id.0 as usize) + } + + /// Returns the process group and session of process `pid`. + fn membership(&self, pid: i32) -> Result { + self.global + .litebox + .process_info(ProcessId(pid.cast_unsigned())) + .map(|info| info.membership) + } + + /// Returns the process group and session of process `pid`, or of this process if `pid` is + /// zero. + fn membership_for_syscall(&self, pid: i32) -> Result { + let pid = match pid { + 0 => self.sys_getpid(), + 1.. => pid, + _ => return Err(Errno::ESRCH), + }; + Ok(self.membership(pid)?) + } + + /// Handle syscall `setpgid`. + /// + /// Unlike Linux, a child that already called `execve` can still be moved. + pub(crate) fn sys_setpgid(&self, pid: i32, process_group: i32) -> Result { + let pid = if pid == 0 { self.sys_getpid() } else { pid }; + let process_group = if process_group == 0 { + pid + } else { + process_group + }; + if process_group < 0 { + return Err(Errno::EINVAL); + } + // A `vfork` child makes its requests as its parent, so it must not reach the parent or + // the parent's other children, which it cannot on Linux. + if pid < 0 || (self.vfork.borrow().is_some() && pid != self.sys_getpid()) { + return Err(Errno::ESRCH); + } + self.global.litebox.set_process_group( + ProcessId(pid.cast_unsigned()), + ProcessId(process_group.cast_unsigned()), + )?; + Ok(0) + } + + /// Handle syscall `getpgid`. + pub(crate) fn sys_getpgid(&self, pid: i32) -> Result { + let membership = self.membership_for_syscall(pid)?; + Ok(membership.process_group.0 as usize) + } + + /// Handle syscall `getpgrp`. + pub(crate) fn sys_getpgrp(&self) -> Result { + self.sys_getpgid(0) + } + + /// Handle syscall `setsid`. + pub(crate) fn sys_setsid(&self) -> Result { + let pid = self.sys_getpid(); + self.global + .litebox + .create_session(ProcessId(pid.cast_unsigned()))?; + Ok(pid.cast_unsigned() as usize) + } + + /// Handle syscall `getsid`. + pub(crate) fn sys_getsid(&self, pid: i32) -> Result { + let membership = self.membership_for_syscall(pid)?; + Ok(membership.session.0 as usize) } /// Handle syscall `getpid`. @@ -2365,8 +2318,21 @@ impl Task { .map_or(self.pid, |state| state.child_pid) } + /// Handle syscall `getppid`. + /// + /// Like Linux, a process without a parent, such as the first process, has parent ID zero. pub(crate) fn sys_getppid(&self) -> i32 { - self.vfork.borrow().as_ref().map_or(self.ppid, |_| self.pid) + if self.vfork.borrow().is_some() { + return self.pid; + } + // Failure means the process service is unavailable or has failed, so the process tree + // cannot be read. + self.global + .litebox + .process_info(ProcessId(self.pid.cast_unsigned())) + .ok() + .and_then(|info| info.parent) + .map_or(0, |parent| parent.0.cast_signed()) } /// Handle syscall `getuid`. @@ -2606,7 +2572,6 @@ impl Task { let umask = u32::from(fs.umask().bits()); drop(fs); let mut startup = LinuxProgramStartup { - parent_process_id: self.pid, uid: self.credentials.uid, euid: self.credentials.euid, gid: self.credentials.gid, diff --git a/litebox_shim_linux/src/syscalls/signal/mod.rs b/litebox_shim_linux/src/syscalls/signal/mod.rs index 7df83371d..a46a7d948 100644 --- a/litebox_shim_linux/src/syscalls/signal/mod.rs +++ b/litebox_shim_linux/src/syscalls/signal/mod.rs @@ -25,6 +25,7 @@ use litebox::process::ProcessError; use litebox::{shim::Exception, sync::Mutex, utils::ReinterpretUnsignedExt as _}; use litebox_broker_protocol::ProcessId; use litebox_broker_protocol::process::ProcessExitStatus; +use litebox_broker_protocol::signal::SignalTarget; use litebox_common_linux::signal::{ CLD_EXITED, CLD_KILLED, FPE_INTDIV, ILL_ILLOPN, MINSIGSTKSZ, NSIG, SI_KERNEL, SI_USER, SIG_DFL, SIG_IGN, SaFlags, SigAction, SigAltStack, SigSet, Siginfo, SiginfoData, SigmaskHow, Signal, @@ -756,10 +757,10 @@ impl Task { }; if signal == Signal::SIGCHLD && act.is_some() { // Linux decides whether to send `SIGCHLD` as the child terminates, so terminations - // already notified to this process are observed under the old action. A termination + // already notified to this process are taken under the old action. A termination // concurrent with this call, or not yet notified, which the guest cannot tell apart, // may still be reaped under the old action and signaled under the new one. - let _ = self.observe_child_terminations(); + self.take_signals(false); } let handlers = self.signals.handlers.borrow(); @@ -825,24 +826,45 @@ impl Task { self.send_signal(signal, siginfo_kill(signal)); } Ok(0) - } else if let (Some(pid @ 1..), None) = (pid, tid) { - let signal = signal.map_or(0, |signal| signal.as_i32().cast_unsigned()); - match self - .global - .litebox - .send_signal(ProcessId(pid.cast_unsigned()), signal) - { - Ok(()) => Ok(0), - // Without a process service, no other process exists. - Err(ProcessError::Unavailable) => Err(Errno::ESRCH), - Err(error) => Err(error.into()), - } + } else if let (Some(pid), None) = (pid, tid) { + self.kill_processes(pid, signal) } else { log_unsupported!("sys_{{t|tg}}kill with remote pid/tid"); Err(Errno::ESRCH) } } + /// Sends `signal` to the processes `kill` selects with `pid`, other than this process alone. + fn kill_processes(&self, pid: i32, signal: Option) -> Result { + let target = match pid { + 1.. => SignalTarget::Process(ProcessId(pid.cast_unsigned())), + 0 => SignalTarget::ProcessGroup( + self.global + .litebox + .process_info(ProcessId(self.pid.cast_unsigned()))? + .membership + .process_group, + ), + -1 => SignalTarget::All, + i32::MIN => return Err(Errno::ESRCH), + _ => SignalTarget::ProcessGroup(ProcessId(pid.unsigned_abs())), + }; + let number = signal.map_or(0, |signal| signal.as_i32().cast_unsigned()); + match self.global.litebox.send_signal(target, number) { + Ok(()) => {} + // Without a process service, no other process exists. + Err(ProcessError::Unavailable) if matches!(target, SignalTarget::Process(_)) => { + return Err(Errno::ESRCH); + } + Err(error) => return Err(error.into()), + } + // The group may include this process. + if signal.is_some() && matches!(target, SignalTarget::ProcessGroup(_)) { + self.take_signals(true); + } + Ok(0) + } + /// Returns whether there are any pending signals that can be delivered. pub(crate) fn has_pending_signals(&self) -> bool { let blocked = self.signals.blocked.get(); diff --git a/litebox_shim_linux/src/wait.rs b/litebox_shim_linux/src/wait.rs index 38d2817d3..91e72fe57 100644 --- a/litebox_shim_linux/src/wait.rs +++ b/litebox_shim_linux/src/wait.rs @@ -51,15 +51,14 @@ impl Task { } /// Queues the signals raised outside this thread: platform signals, the fallback alarm's - /// `SIGALRM`, `SIGCHLD` for terminated children, and signals other processes sent. + /// `SIGALRM`, signals other processes sent, and `SIGCHLD` for terminated children. fn queue_async_signals(&self) { self.global.platform.take_pending_signals(|signal| { self.queue_signals(signal); }); #[cfg(feature = "alarm_fallback")] self.check_alarm_deadline(); - self.check_for_child_terminations(); - self.take_signals(); + self.take_signals(false); } }