From b8ad93e3a5429ffa749758846fe714df2f98879d Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sun, 30 Aug 2026 00:35:41 -0700 Subject: [PATCH 1/3] Add tests for post-deploy failures deleting live folded-run rows deployAdoptedWorkflowFromSource resolving means a live child now exists on the sidecar; a plain Error out of markRunDeployClone or sendRunGrants afterward currently reads as "nothing was deployed" and deletes the run's rows, orphaning a live, unauthorized agent with no DB trace (CL-7213). --- packages/folded-runs/test/launch.test.ts | 122 ++++++++++++++++++++++- 1 file changed, 121 insertions(+), 1 deletion(-) diff --git a/packages/folded-runs/test/launch.test.ts b/packages/folded-runs/test/launch.test.ts index 1cf240ce2..fa09651df 100644 --- a/packages/folded-runs/test/launch.test.ts +++ b/packages/folded-runs/test/launch.test.ts @@ -14,7 +14,12 @@ // a controllable stub — spreading through every other export unchanged // — so a real tenant catalog is never required to prove the wiring. import { describe, expect, mock, test } from "bun:test"; -import { agentSession, principal, workflowRun } from "@intx/db/schema"; +import { + agentSession, + principal, + workflowDefinition, + workflowRun, +} from "@intx/db/schema"; import { foldedRun } from "../src/schema"; import { SessionLaunchError } from "@intx/hub-sessions"; import type { EventCollectorRegistry, SidecarRouter } from "@intx/hub-sessions"; @@ -938,6 +943,121 @@ describe("launchFoldedRun", () => { expect(runUpdate?.values).toEqual({ status: "failed" }); }); + // `deployAdoptedWorkflowFromSource` resolving means a live child now + // exists on the sidecar; `sendRunGrants` returning `false` (address not + // yet routable — a real hub-link drop between the deploy ack and the + // grants send) must never read as "nothing was deployed." CL-7213: this + // used to throw a plain `Error`, which the catch's `leaked` check missed + // entirely, deleting the row for a run that was actually live and now + // permanently unauthorized and untracked. + test("marks the run failed (not deleted) when sendRunGrants fails after a successful deploy", async () => { + resolveDefinitionSourcesResult = { + ok: true, + sources: [ + { + id: "off_1", + provider: "anthropic", + baseURL: "https://inference.invalid", + apiKey: "placeholder", + model: "claude-sonnet-5", + }, + ], + defaultSource: "off_1", + }; + + const db = createFakeDb(); + const sessionService = createFakeSessionService(); + const eventCollectors = createFakeEventCollectors(); + + await expect( + launchFoldedRun( + { + db: db as never, + sessionService, + assetService: createFakeAssetService(), + sidecarRouter: createFakeSidecarRouter(false), + toolGrantsForPins: () => [], + eventCollectors, + }, + { + tenantId: "ten_1", + instanceId: "ins_workbench1", + triggerAddress: "ins_workbench1@ten1.workbench.test", + definitionId: "wfd_workbench1", + foldedBody: FOLDED_BODY, + launchLabel: "the workbench host", + }, + ), + ).rejects.toThrow(SessionLaunchError); + + expect(db.deleted).toEqual([]); + const runUpdate = db.updated.find((row) => row.table === workflowRun); + expect(runUpdate?.values).toEqual({ status: "failed" }); + }); + + // Same class of bug as the `sendRunGrants` case above, but for the + // other post-deploy step: `markRunDeployClone` runs after the live + // deploy resolves too, so a plain `Error` out of it must be treated as + // a leaked deploy for the same reason. + test("marks the run failed (not deleted) when markRunDeployClone fails after a successful deploy", async () => { + resolveDefinitionSourcesResult = { + ok: true, + sources: [ + { + id: "off_1", + provider: "anthropic", + baseURL: "https://inference.invalid", + apiKey: "placeholder", + model: "claude-sonnet-5", + }, + ], + defaultSource: "off_1", + }; + + // A second definition read that differs from the first is what makes + // `markRunDeployClone` attempt the `workflowDefinition` update at all + // (see `createFakeDb`'s own doc) — that update is the call this test + // makes fail. + const db = createFakeDb("ast_definition1", [ + "wfd_definition1", + "wfd_definition1_clone", + ]); + const originalUpdate = db.update.bind(db); + db.update = ((table: unknown) => { + if (table === workflowDefinition) { + throw new Error("clone marking failed"); + } + return originalUpdate(table); + }) as typeof db.update; + const sessionService = createFakeSessionService(); + const eventCollectors = createFakeEventCollectors(); + + await expect( + launchFoldedRun( + { + db: db as never, + sessionService, + assetService: createFakeAssetService(), + sidecarRouter: createFakeSidecarRouter(), + toolGrantsForPins: () => [], + eventCollectors, + }, + { + tenantId: "ten_1", + instanceId: "ins_workbench1", + triggerAddress: "ins_workbench1@ten1.workbench.test", + definitionId: "wfd_workbench1", + foldedBody: FOLDED_BODY, + launchLabel: "the workbench host", + }, + ), + ).rejects.toThrow(SessionLaunchError); + + expect(db.deleted).toEqual([]); + const runUpdate = db.updated.find((row) => row.table === workflowRun); + expect(runUpdate?.values).toEqual({ status: "failed" }); + }); + test("throws InferenceResolutionError when the tenant catalog has no launchable source", async () => { resolveDefinitionSourcesResult = { ok: false, From b25221de213d7e2acebc05c285207e88fb8a0e80 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sun, 30 Aug 2026 04:16:34 -0700 Subject: [PATCH 2/3] Wrap post-deploy grants/clone failures as leaked SessionLaunchError deployAdoptedWorkflowFromSource resolving means a live child now exists on the sidecar. A plain Error out of markRunDeployClone or sendRunGrants afterward used to read as "nothing was deployed," so launchFoldedRun's failure-path rollback deleted the run's rows, leaving a live agent running, unauthorized, and untracked (CL-7213). Both failure sites now throw SessionLaunchError(..., leakedAgent: true), matching the existing signal the rollback already branches on. --- packages/folded-runs/src/launch.ts | 66 ++++++++++++++++++------------ 1 file changed, 40 insertions(+), 26 deletions(-) diff --git a/packages/folded-runs/src/launch.ts b/packages/folded-runs/src/launch.ts index ccc4d10fa..bd8a523b4 100644 --- a/packages/folded-runs/src/launch.ts +++ b/packages/folded-runs/src/launch.ts @@ -641,34 +641,48 @@ export async function deployAtHead( : {}), }); - await markRunDeployClone( - deps.db, - params.instanceId, - definitionIdBeforeDeploy, - ); - - // Produce the run's `run.grants` frame, the same contract upstream's hub - // fires on every run birth: the sidecar writes it to - // `runs//grants.json` in the deployment's workflow-run repo, and - // the supervisor's `onRunStart` barrier reads that file to authorize the - // run. A folded run is self-anchored, so its run id IS its deployment id, - // and `grants` — the tool-pin and credential-binding set already deployed - // as `config.grants` — is the run's whole grant set. - // - // Sent after the deploy resolves and before any trigger mail: the sidecar - // registers its grants handler during the deploy the ack acknowledges, and - // both frames ride the same per-address channel, so same-socket FIFO puts - // the grants on disk ahead of the mail that starts the run. - if ( - !deps.sidecarRouter.sendRunGrants( - params.triggerAddress, + // Everything from here on runs AFTER `deployAdoptedWorkflowFromSource` + // resolved, meaning a live child now exists on the sidecar. A failure + // in this region is a leaked deploy, never a "nothing happened" — + // wrapped as `SessionLaunchError(..., leakedAgent: true)` so a caller's + // failure-path rollback (`launchFoldedRun`) preserves the run's rows + // instead of deleting them out from under a genuinely running agent + // (CL-7213). The distinction is structural (this try/catch boundary), + // not a classification any caller has to reconstruct from error type. + try { + await markRunDeployClone( + deps.db, params.instanceId, - grants, - ) - ) { - throw new Error( - `${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`, + definitionIdBeforeDeploy, ); + + // Produce the run's `run.grants` frame, the same contract upstream's + // hub fires on every run birth: the sidecar writes it to + // `runs//grants.json` in the deployment's workflow-run repo, + // and the supervisor's `onRunStart` barrier reads that file to + // authorize the run. A folded run is self-anchored, so its run id IS + // its deployment id, and `grants` — the tool-pin and + // credential-binding set already deployed as `config.grants` — is + // the run's whole grant set. + // + // Sent after the deploy resolves and before any trigger mail: the + // sidecar registers its grants handler during the deploy the ack + // acknowledges, and both frames ride the same per-address channel, + // so same-socket FIFO puts the grants on disk ahead of the mail that + // starts the run. + if ( + !deps.sidecarRouter.sendRunGrants( + params.triggerAddress, + params.instanceId, + grants, + ) + ) { + throw new Error( + `${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`, + ); + } + } catch (err) { + throw new SessionLaunchError("grants", err, true); } return { sourcesDigest: inferenceSourcesDigest(resolution) }; } From bb8b516f04bbba06edaaef0267d3b83eed2a695d Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Sun, 30 Aug 2026 06:08:28 -0700 Subject: [PATCH 3/3] Give markRunDeployClone and sendRunGrants distinct SessionLaunchError phases Both post-deploy failure sites threw SessionLaunchError with the same "grants" phase; SessionLaunchError.phase is surfaced in upstream cleanup diagnostics, so a markRunDeployClone failure was misreported as a grants failure. Each step now throws with its own phase label. --- packages/folded-runs/src/launch.ts | 67 +++++++++++++++++------------- 1 file changed, 38 insertions(+), 29 deletions(-) diff --git a/packages/folded-runs/src/launch.ts b/packages/folded-runs/src/launch.ts index bd8a523b4..738cba373 100644 --- a/packages/folded-runs/src/launch.ts +++ b/packages/folded-runs/src/launch.ts @@ -643,46 +643,55 @@ export async function deployAtHead( // Everything from here on runs AFTER `deployAdoptedWorkflowFromSource` // resolved, meaning a live child now exists on the sidecar. A failure - // in this region is a leaked deploy, never a "nothing happened" — + // in either step below is a leaked deploy, never a "nothing happened" — // wrapped as `SessionLaunchError(..., leakedAgent: true)` so a caller's // failure-path rollback (`launchFoldedRun`) preserves the run's rows // instead of deleting them out from under a genuinely running agent - // (CL-7213). The distinction is structural (this try/catch boundary), - // not a classification any caller has to reconstruct from error type. + // (CL-7213). The distinction is structural (everything past this + // comment), not a classification any caller has to reconstruct from + // error type. Each step gets its own phase label — `SessionLaunchError.phase` + // is surfaced in cleanup diagnostics upstream (see + // `deployAgentUndeploy` failure logging in `session-service.ts`), so a + // shared label across both steps would misreport which one actually + // failed. try { await markRunDeployClone( deps.db, params.instanceId, definitionIdBeforeDeploy, ); + } catch (err) { + throw new SessionLaunchError("clone", err, true); + } - // Produce the run's `run.grants` frame, the same contract upstream's - // hub fires on every run birth: the sidecar writes it to - // `runs//grants.json` in the deployment's workflow-run repo, - // and the supervisor's `onRunStart` barrier reads that file to - // authorize the run. A folded run is self-anchored, so its run id IS - // its deployment id, and `grants` — the tool-pin and - // credential-binding set already deployed as `config.grants` — is - // the run's whole grant set. - // - // Sent after the deploy resolves and before any trigger mail: the - // sidecar registers its grants handler during the deploy the ack - // acknowledges, and both frames ride the same per-address channel, - // so same-socket FIFO puts the grants on disk ahead of the mail that - // starts the run. - if ( - !deps.sidecarRouter.sendRunGrants( - params.triggerAddress, - params.instanceId, - grants, - ) - ) { - throw new Error( + // Produce the run's `run.grants` frame, the same contract upstream's + // hub fires on every run birth: the sidecar writes it to + // `runs//grants.json` in the deployment's workflow-run repo, + // and the supervisor's `onRunStart` barrier reads that file to + // authorize the run. A folded run is self-anchored, so its run id IS + // its deployment id, and `grants` — the tool-pin and + // credential-binding set already deployed as `config.grants` — is + // the run's whole grant set. + // + // Sent after the deploy resolves and before any trigger mail: the + // sidecar registers its grants handler during the deploy the ack + // acknowledges, and both frames ride the same per-address channel, + // so same-socket FIFO puts the grants on disk ahead of the mail that + // starts the run. + if ( + !deps.sidecarRouter.sendRunGrants( + params.triggerAddress, + params.instanceId, + grants, + ) + ) { + throw new SessionLaunchError( + "grants", + new Error( `${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`, - ); - } - } catch (err) { - throw new SessionLaunchError("grants", err, true); + ), + true, + ); } return { sourcesDigest: inferenceSourcesDigest(resolution) }; }