Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
59 changes: 41 additions & 18 deletions packages/folded-runs/src/launch.ts
Original file line number Diff line number Diff line change
Expand Up @@ -641,33 +641,56 @@ export async function deployAtHead(
: {}),
});

await markRunDeployClone(
deps.db,
params.instanceId,
definitionIdBeforeDeploy,
);
// Everything from here on runs AFTER `deployAdoptedWorkflowFromSource`
// resolved, meaning a live child now exists on the sidecar. A failure
// in either step below is a leaked deploy, never a "nothing happened" —
// wrapped as `SessionLaunchError(..., leakedAgent: true)` so a caller's
// failure-path rollback (`launchFoldedRun`) preserves the run's rows
// instead of deleting them out from under a genuinely running agent
// (CL-7213). The distinction is structural (everything past this
// comment), not a classification any caller has to reconstruct from
// error type. Each step gets its own phase label — `SessionLaunchError.phase`
// is surfaced in cleanup diagnostics upstream (see
// `deployAgentUndeploy` failure logging in `session-service.ts`), so a
// shared label across both steps would misreport which one actually
// failed.
try {
await markRunDeployClone(
deps.db,
params.instanceId,
definitionIdBeforeDeploy,
);
} catch (err) {
throw new SessionLaunchError("clone", err, true);
}

// Produce the run's `run.grants` frame, the same contract upstream's hub
// fires on every run birth: the sidecar writes it to
// `runs/<runId>/grants.json` in the deployment's workflow-run repo, and
// the supervisor's `onRunStart` barrier reads that file to authorize the
// run. A folded run is self-anchored, so its run id IS its deployment id,
// and `grants` — the tool-pin and credential-binding set already deployed
// as `config.grants` — is the run's whole grant set.
// Produce the run's `run.grants` frame, the same contract upstream's
// hub fires on every run birth: the sidecar writes it to
// `runs/<runId>/grants.json` in the deployment's workflow-run repo,
// and the supervisor's `onRunStart` barrier reads that file to
// authorize the run. A folded run is self-anchored, so its run id IS
// its deployment id, and `grants` — the tool-pin and
// credential-binding set already deployed as `config.grants` — is
// the run's whole grant set.
//
// Sent after the deploy resolves and before any trigger mail: the sidecar
// registers its grants handler during the deploy the ack acknowledges, and
// both frames ride the same per-address channel, so same-socket FIFO puts
// the grants on disk ahead of the mail that starts the run.
// Sent after the deploy resolves and before any trigger mail: the
// sidecar registers its grants handler during the deploy the ack
// acknowledges, and both frames ride the same per-address channel,
// so same-socket FIFO puts the grants on disk ahead of the mail that
// starts the run.
if (
!deps.sidecarRouter.sendRunGrants(
params.triggerAddress,
params.instanceId,
grants,
)
) {
throw new Error(
`${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`,
throw new SessionLaunchError(
"grants",
new Error(
`${params.launchLabel}: deployment ${params.triggerAddress} is not routable for run ${params.instanceId}; cannot deliver its run grants, so the run would start under-authorized`,
),
true,
);
}
return { sourcesDigest: inferenceSourcesDigest(resolution) };
Expand Down
122 changes: 121 additions & 1 deletion packages/folded-runs/test/launch.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,12 @@
// a controllable stub — spreading through every other export unchanged
// — so a real tenant catalog is never required to prove the wiring.
import { describe, expect, mock, test } from "bun:test";
import { agentSession, principal, workflowRun } from "@intx/db/schema";
import {
agentSession,
principal,
workflowDefinition,
workflowRun,
} from "@intx/db/schema";
import { foldedRun } from "../src/schema";
import { SessionLaunchError } from "@intx/hub-sessions";
import type { EventCollectorRegistry, SidecarRouter } from "@intx/hub-sessions";
Expand Down Expand Up @@ -938,6 +943,121 @@ describe("launchFoldedRun", () => {
expect(runUpdate?.values).toEqual({ status: "failed" });
});

// `deployAdoptedWorkflowFromSource` resolving means a live child now
// exists on the sidecar; `sendRunGrants` returning `false` (address not
// yet routable — a real hub-link drop between the deploy ack and the
// grants send) must never read as "nothing was deployed." CL-7213: this
// used to throw a plain `Error`, which the catch's `leaked` check missed
// entirely, deleting the row for a run that was actually live and now
// permanently unauthorized and untracked.
test("marks the run failed (not deleted) when sendRunGrants fails after a successful deploy", async () => {
resolveDefinitionSourcesResult = {
ok: true,
sources: [
{
id: "off_1",
provider: "anthropic",
baseURL: "https://inference.invalid",
apiKey: "placeholder",
model: "claude-sonnet-5",
},
],
defaultSource: "off_1",
};

const db = createFakeDb();
const sessionService = createFakeSessionService();
const eventCollectors = createFakeEventCollectors();

await expect(
launchFoldedRun(
{
db: db as never,
sessionService,
assetService: createFakeAssetService(),
sidecarRouter: createFakeSidecarRouter(false),
toolGrantsForPins: () => [],
eventCollectors,
},
{
tenantId: "ten_1",
instanceId: "ins_workbench1",
triggerAddress: "ins_workbench1@ten1.workbench.test",
definitionId: "wfd_workbench1",
foldedBody: FOLDED_BODY,
launchLabel: "the workbench host",
},
),
).rejects.toThrow(SessionLaunchError);

expect(db.deleted).toEqual([]);
const runUpdate = db.updated.find((row) => row.table === workflowRun);
expect(runUpdate?.values).toEqual({ status: "failed" });
});

// Same class of bug as the `sendRunGrants` case above, but for the
// other post-deploy step: `markRunDeployClone` runs after the live
// deploy resolves too, so a plain `Error` out of it must be treated as
// a leaked deploy for the same reason.
test("marks the run failed (not deleted) when markRunDeployClone fails after a successful deploy", async () => {
resolveDefinitionSourcesResult = {
ok: true,
sources: [
{
id: "off_1",
provider: "anthropic",
baseURL: "https://inference.invalid",
apiKey: "placeholder",
model: "claude-sonnet-5",
},
],
defaultSource: "off_1",
};

// A second definition read that differs from the first is what makes
// `markRunDeployClone` attempt the `workflowDefinition` update at all
// (see `createFakeDb`'s own doc) — that update is the call this test
// makes fail.
const db = createFakeDb("ast_definition1", [
"wfd_definition1",
"wfd_definition1_clone",
]);
const originalUpdate = db.update.bind(db);
db.update = ((table: unknown) => {
if (table === workflowDefinition) {
throw new Error("clone marking failed");
}
return originalUpdate(table);
}) as typeof db.update;
const sessionService = createFakeSessionService();
const eventCollectors = createFakeEventCollectors();

await expect(
launchFoldedRun(
{
db: db as never,
sessionService,
assetService: createFakeAssetService(),
sidecarRouter: createFakeSidecarRouter(),
toolGrantsForPins: () => [],
eventCollectors,
},
{
tenantId: "ten_1",
instanceId: "ins_workbench1",
triggerAddress: "ins_workbench1@ten1.workbench.test",
definitionId: "wfd_workbench1",
foldedBody: FOLDED_BODY,
launchLabel: "the workbench host",
},
),
).rejects.toThrow(SessionLaunchError);

expect(db.deleted).toEqual([]);
const runUpdate = db.updated.find((row) => row.table === workflowRun);
expect(runUpdate?.values).toEqual({ status: "failed" });
});

test("throws InferenceResolutionError when the tenant catalog has no launchable source", async () => {
resolveDefinitionSourcesResult = {
ok: false,
Expand Down
Loading