From b11c4cebb6b4d3225c22362354ed52f2ee35be2e Mon Sep 17 00:00:00 2001 From: Sree Narayanan Date: Thu, 8 Oct 2026 20:16:51 +0400 Subject: [PATCH 1/2] fix(agentos): wait for a spawned process to start before reading its output --- crates/client/src/process.rs | 27 ++++++++++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/crates/client/src/process.rs b/crates/client/src/process.rs index f7d966e476..c74348960f 100644 --- a/crates/client/src/process.rs +++ b/crates/client/src/process.rs @@ -1000,7 +1000,14 @@ impl AgentOs { max_events: Option, max_bytes: Option, ) -> std::result::Result { - let (process_id, retain_output, execution_id, execution_generation) = self + let ( + process_id, + retain_output, + execution_id, + execution_generation, + mut started, + mut outcome, + ) = self .inner() .processes .read(&pid, |_, entry| { @@ -1009,9 +1016,27 @@ impl AgentOs { entry.retain_output, entry.execution_id.clone(), entry.execution_generation, + entry.kernel_pid.subscribe(), + entry.exit_tx.subscribe(), ) }) .ok_or(ClientError::ProcessNotFound(pid))?; + // `spawn_process` returns before the sidecar has the process. Wait until the Execute + // request lands or fails, so a read never reaches the sidecar before the process exists. + loop { + if started.borrow().is_some() { + break; + } + match &*outcome.borrow() { + ProcessOutcome::Failed { error, .. } => return Err(error.clone()), + ProcessOutcome::Exited(_) => break, + ProcessOutcome::Pending => {} + } + tokio::select! { + changed = started.changed() => if changed.is_err() { break }, + changed = outcome.changed() => if changed.is_err() { break }, + } + } if !retain_output { return Err(ClientError::Sidecar(format!( "process {pid} was not spawned with output retention enabled" From d84d3a260574e2b5bc4bd95291d60d14871db7b9 Mon Sep 17 00:00:00 2001 From: Sree Narayanan Date: Thu, 8 Oct 2026 23:02:15 +0400 Subject: [PATCH 2/2] test(agentos): cover reading output right after a failed spawn --- crates/client/tests/process_e2e.rs | 34 ++++++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/crates/client/tests/process_e2e.rs b/crates/client/tests/process_e2e.rs index efc14f0c85..86a7d9f805 100644 --- a/crates/client/tests/process_e2e.rs +++ b/crates/client/tests/process_e2e.rs @@ -120,6 +120,40 @@ async fn spawn_rejection_and_fast_exit_remain_distinct_for_late_waiters() { .expect("process outcome parity"); } +#[tokio::test] +async fn output_read_right_after_a_failed_spawn_returns_the_launch_error() { + if !common::require_sidecar("output_read_right_after_a_failed_spawn_returns_the_launch_error") { + return; + } + let os = common::new_vm().await; + let result = tokio::time::timeout(std::time::Duration::from_secs(20), async { + let launch = os.spawn_process( + "agentos-nonexistent-review-command", + Vec::new(), + SpawnOptions { + retain_output: true, + ..Default::default() + }, + )?; + // Read before the launch has settled: the caller must get the launch error, not a + // missing-process error from asking the sidecar before it knows the process. + let read = os.read_process_output(launch.pid, None, None, None).await; + anyhow::ensure!( + matches!(&read, Err(ClientError::Kernel { code, .. }) if code == "ENOENT"), + "an output read right after a failed spawn must return the launch error: {read:?}" + ); + Ok::<_, anyhow::Error>(()) + }) + .await; + tokio::time::timeout(std::time::Duration::from_secs(10), os.shutdown()) + .await + .expect("bounded VM cleanup") + .expect("shutdown VM"); + result + .expect("output read finishes within the timeout") + .expect("output read returns the launch error"); +} + #[tokio::test] async fn exec_argv_timeout_confirms_node_guest_exit() { if !common::require_sidecar("exec_argv_timeout_confirms_node_guest_exit") {