diff --git a/packages/folded-runs/src/launch.ts b/packages/folded-runs/src/launch.ts index ccc4d10fa..738cba373 100644 --- a/packages/folded-runs/src/launch.ts +++ b/packages/folded-runs/src/launch.ts @@ -641,24 +641,43 @@ export async function deployAtHead( : {}), }); - await markRunDeployClone( - deps.db, - params.instanceId, - definitionIdBeforeDeploy, - ); + // Everything from here on runs AFTER `deployAdoptedWorkflowFromSource` + // resolved, meaning a live child now exists on the sidecar. A failure + // in either step below is a leaked deploy, never a "nothing happened" — + // wrapped as `SessionLaunchError(..., leakedAgent: true)` so a caller's + // failure-path rollback (`launchFoldedRun`) preserves the run's rows + // instead of deleting them out from under a genuinely running agent + // (CL-7213). The distinction is structural (everything past this + // comment), not a classification any caller has to reconstruct from + // error type. Each step gets its own phase label — `SessionLaunchError.phase` + // is surfaced in cleanup diagnostics upstream (see + // `deployAgentUndeploy` failure logging in `session-service.ts`), so a + // shared label across both steps would misreport which one actually + // failed. + try { + await markRunDeployClone( + deps.db, + params.instanceId, + definitionIdBeforeDeploy, + ); + } catch (err) { + throw new SessionLaunchError("clone", err, true); + } - // Produce the run's `run.grants` frame, the same contract upstream's hub - // fires on every run birth: the sidecar writes it to - // `runs//grants.json` in the deployment's workflow-run repo, and - // the supervisor's `onRunStart` barrier reads that file to authorize the - // run. A folded run is self-anchored, so its run id IS its deployment id, - // and `grants` — the tool-pin and credential-binding set already deployed - // as `config.grants` — is the run's whole grant set. + // Produce the run's `run.grants` frame, the same contract upstream's + // hub fires on every run birth: the sidecar writes it to + // `runs//grants.json` in the deployment's workflow-run repo, + // and the supervisor's `onRunStart` barrier reads that file to + // authorize the run. A folded run is self-anchored, so its run id IS + // its deployment id, and `grants` — the tool-pin and + // credential-binding set already deployed as `config.grants` — is + // the run's whole grant set. // - // Sent after the deploy resolves and before any trigger mail: the sidecar - // registers its grants handler during the deploy the ack acknowledges, and - // both frames ride the same per-address channel, so same-socket FIFO puts - // the grants on disk ahead of the mail that starts the run. + // Sent after the deploy resolves and before any trigger mail: the + // sidecar registers its grants handler during the deploy the ack + // acknowledges, and both frames ride the same per-address channel, + // so same-socket FIFO puts the grants on disk ahead of the mail that + // starts the run. if ( !deps.sidecarRouter.sendRunGrants( params.triggerAddress, @@ -666,8 +685,12 @@ export async function deployAtHead( grants, ) ) { - throw new Error( - `${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`, + throw new SessionLaunchError( + "grants", + new Error( + `${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`, + ), + true, ); } return { sourcesDigest: inferenceSourcesDigest(resolution) }; diff --git a/packages/folded-runs/test/launch.test.ts b/packages/folded-runs/test/launch.test.ts index 1cf240ce2..fa09651df 100644 --- a/packages/folded-runs/test/launch.test.ts +++ b/packages/folded-runs/test/launch.test.ts @@ -14,7 +14,12 @@ // a controllable stub — spreading through every other export unchanged // — so a real tenant catalog is never required to prove the wiring. import { describe, expect, mock, test } from "bun:test"; -import { agentSession, principal, workflowRun } from "@intx/db/schema"; +import { + agentSession, + principal, + workflowDefinition, + workflowRun, +} from "@intx/db/schema"; import { foldedRun } from "../src/schema"; import { SessionLaunchError } from "@intx/hub-sessions"; import type { EventCollectorRegistry, SidecarRouter } from "@intx/hub-sessions"; @@ -938,6 +943,121 @@ describe("launchFoldedRun", () => { expect(runUpdate?.values).toEqual({ status: "failed" }); }); + // `deployAdoptedWorkflowFromSource` resolving means a live child now + // exists on the sidecar; `sendRunGrants` returning `false` (address not + // yet routable — a real hub-link drop between the deploy ack and the + // grants send) must never read as "nothing was deployed." CL-7213: this + // used to throw a plain `Error`, which the catch's `leaked` check missed + // entirely, deleting the row for a run that was actually live and now + // permanently unauthorized and untracked. + test("marks the run failed (not deleted) when sendRunGrants fails after a successful deploy", async () => { + resolveDefinitionSourcesResult = { + ok: true, + sources: [ + { + id: "off_1", + provider: "anthropic", + baseURL: "https://inference.invalid", + apiKey: "placeholder", + model: "claude-sonnet-5", + }, + ], + defaultSource: "off_1", + }; + + const db = createFakeDb(); + const sessionService = createFakeSessionService(); + const eventCollectors = createFakeEventCollectors(); + + await expect( + launchFoldedRun( + { + db: db as never, + sessionService, + assetService: createFakeAssetService(), + sidecarRouter: createFakeSidecarRouter(false), + toolGrantsForPins: () => [], + eventCollectors, + }, + { + tenantId: "ten_1", + instanceId: "ins_workbench1", + triggerAddress: "ins_workbench1@ten1.workbench.test", + definitionId: "wfd_workbench1", + foldedBody: FOLDED_BODY, + launchLabel: "the workbench host", + }, + ), + ).rejects.toThrow(SessionLaunchError); + + expect(db.deleted).toEqual([]); + const runUpdate = db.updated.find((row) => row.table === workflowRun); + expect(runUpdate?.values).toEqual({ status: "failed" }); + }); + + // Same class of bug as the `sendRunGrants` case above, but for the + // other post-deploy step: `markRunDeployClone` runs after the live + // deploy resolves too, so a plain `Error` out of it must be treated as + // a leaked deploy for the same reason. + test("marks the run failed (not deleted) when markRunDeployClone fails after a successful deploy", async () => { + resolveDefinitionSourcesResult = { + ok: true, + sources: [ + { + id: "off_1", + provider: "anthropic", + baseURL: "https://inference.invalid", + apiKey: "placeholder", + model: "claude-sonnet-5", + }, + ], + defaultSource: "off_1", + }; + + // A second definition read that differs from the first is what makes + // `markRunDeployClone` attempt the `workflowDefinition` update at all + // (see `createFakeDb`'s own doc) — that update is the call this test + // makes fail. + const db = createFakeDb("ast_definition1", [ + "wfd_definition1", + "wfd_definition1_clone", + ]); + const originalUpdate = db.update.bind(db); + db.update = ((table: unknown) => { + if (table === workflowDefinition) { + throw new Error("clone marking failed"); + } + return originalUpdate(table); + }) as typeof db.update; + const sessionService = createFakeSessionService(); + const eventCollectors = createFakeEventCollectors(); + + await expect( + launchFoldedRun( + { + db: db as never, + sessionService, + assetService: createFakeAssetService(), + sidecarRouter: createFakeSidecarRouter(), + toolGrantsForPins: () => [], + eventCollectors, + }, + { + tenantId: "ten_1", + instanceId: "ins_workbench1", + triggerAddress: "ins_workbench1@ten1.workbench.test", + definitionId: "wfd_workbench1", + foldedBody: FOLDED_BODY, + launchLabel: "the workbench host", + }, + ), + ).rejects.toThrow(SessionLaunchError); + + expect(db.deleted).toEqual([]); + const runUpdate = db.updated.find((row) => row.table === workflowRun); + expect(runUpdate?.values).toEqual({ status: "failed" }); + }); + test("throws InferenceResolutionError when the tenant catalog has no launchable source", async () => { resolveDefinitionSourcesResult = { ok: false,