Skip to content

Commit 64542d7

Browse files
Merge pull request #498 from corbitsdev/cl-6652-drop-the-10-minute-shell-timeout-ceiling-and-run-capability
Honor requested shell timeouts and run evals in parallel
2 parents 3bdda57 + e40824c commit 64542d7

17 files changed

Lines changed: 518 additions & 57 deletions

‎CHANGELOG.md‎

Lines changed: 23 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -11,6 +11,29 @@ matching `## [X.Y.Z]` section (plus install instructions). Do not maintain
1111
parallel copies under `docs/` or `scripts/notes/`. At cut time: rename
1212
`## [Unreleased]` to `## [X.Y.Z] - YYYY-MM-DD`, then run the release script.
1313

14+
## [Unreleased]
15+
16+
### Plugins
17+
18+
- **Requested `run_shell` timeouts are no longer capped at 10 minutes.** The 15s
19+
default when timeout is omitted is unchanged. `shell.maxTimeoutMs` still
20+
clamps the command when set.
21+
22+
- Capability evals accept `--concurrency <n>` (env `CORBITS_EVAL_CONCURRENCY`,
23+
default 1); overlapping `httpFixture` cells isolate `EVAL_HTTP_URL` so
24+
parallel web-bait runs do not share a process.env origin.
25+
26+
### TUI
27+
28+
- **Tool `run()` no longer has an implicit 11-minute wall-clock abort.** The
29+
outer watchdog arms only when Settings set `tools.timeoutMs` /
30+
`tools.maxTimeoutMs`, or when `run_shell` passes a positive `timeout`
31+
(requested plus slack, so this layer cannot beat shell-guard). Unset
32+
settings leave `task` and other tools unbounded; parent cancel, maxTurns,
33+
and eval `--agent-timeout-ms` still bound the run. `tools.maxTimeoutMs`
34+
still clamps non-shell tools when set and does not cap a longer requested
35+
`run_shell`.
36+
1437
## [0.2.99] - 2026-08-21
1538

1639
Skywalker is the primary orchestrator over a closed director fleet: product write tools stay off the primary, and you cannot spawn Skywalker as a task leaf. Workers are not done until they return the four-heading report. First-party action skills ship as slashes; eval runners require an explicit provider/model pair; the style skill no longer refuses non-git folders.

‎evals/capability/README.md‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -117,6 +117,9 @@ bun run eval:capability -- \
117117
--matrix "xai:grok-4.5,openai:gpt-4.1" \
118118
--out evals/capability/results/matrix.json
119119

120+
# Faster live matrix (independent cells; default is serial)
121+
bun run eval:capability -- --provider <name> --model <id> --concurrency 4
122+
120123
# Labeled variants
121124
bun run eval:capability -- --matrix "fast=xai:grok-4.5,strong=openai:gpt-4.1"
122125

@@ -165,6 +168,7 @@ Flags:
165168
| `--agent-timeout-ms <n>` | Wall-clock limit for `runExec` (default `600000`, env `CORBITS_EVAL_AGENT_TIMEOUT_MS`) |
166169
| `--verify-timeout-ms <n>` | Wall-clock limit for `verify.sh` (default `120000`, env `CORBITS_EVAL_VERIFY_TIMEOUT_MS`) |
167170
| `--repeats <n>` | Runs per case×variant cell (default `1`; gate runs use `5`, baseline freezes `3`). Results record every repeat plus per-cell aggregates |
171+
| `--concurrency <n>` | Independent case×variant×repeat cells in parallel (default `1`, env `CORBITS_EVAL_CONCURRENCY`). Each cell still uses its own temp workdir. Use `--concurrency 4` (or similar) to run a live matrix faster |
168172
| `--dry-run` | Load cases × variants and print plan; no inference. Still requires `--provider`/`--model` or `--matrix` |
169173

170174
## Case format

‎evals/capability/lib.test.ts‎

Lines changed: 8 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -20,6 +20,7 @@ import {
2020
baitReproduces,
2121
httpFixtureEnv,
2222
withEnv,
23+
evalHttpEnvGet,
2324
detectProviderFallback,
2425
formatProviderFallback,
2526
resolveRequestedProviderModel,
@@ -811,24 +812,29 @@ describe("withEnv / httpFixtureEnv", () => {
811812

812813
test("makes the fixture origin visible to in-process code the way ssrf-guard reads it", async () => {
813814
const fixture = { url: "http://127.0.0.1:54321/", token: "tok" };
815+
expect(evalHttpEnvGet("EVAL_HTTP_URL")).toBeUndefined();
814816
expect(process.env.EVAL_HTTP_URL).toBeUndefined();
815817
let seenDuring: string | undefined;
816818
await withEnv(httpFixtureEnv(fixture), async () => {
817-
seenDuring = process.env.EVAL_HTTP_URL;
819+
seenDuring = evalHttpEnvGet("EVAL_HTTP_URL");
820+
expect(process.env.EVAL_HTTP_URL).toBeUndefined();
818821
});
819822
expect(seenDuring).toBe(fixture.url);
823+
expect(evalHttpEnvGet("EVAL_HTTP_URL")).toBeUndefined();
820824
expect(process.env.EVAL_HTTP_URL).toBeUndefined();
821825
});
822826

823-
test("restores prior value on throw", async () => {
827+
test("overlay does not leak after throw and leaves process.env untouched", async () => {
824828
process.env.EVAL_HTTP_URL = "http://pre-existing/";
825829
try {
826830
await expect(
827831
withEnv({ EVAL_HTTP_URL: "http://127.0.0.1:1/" }, async () => {
832+
expect(evalHttpEnvGet("EVAL_HTTP_URL")).toBe("http://127.0.0.1:1/");
828833
throw new Error("boom");
829834
}),
830835
).rejects.toThrow("boom");
831836
expect(process.env.EVAL_HTTP_URL).toBe("http://pre-existing/");
837+
expect(evalHttpEnvGet("EVAL_HTTP_URL")).toBe("http://pre-existing/");
832838
} finally {
833839
delete process.env.EVAL_HTTP_URL;
834840
}

‎evals/capability/lib.ts‎

Lines changed: 10 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -5,6 +5,7 @@
55

66
import { readdir, readFile, stat } from "node:fs/promises";
77
import { join, resolve } from "node:path";
8+
import { runWithEvalHttpEnv, evalHttpEnvGet } from "../../src/tools/eval-http-env.js";
89
import {
910
isNumericBehaviorMetric,
1011
parseBehaviorMetrics,
@@ -422,6 +423,8 @@ export function makeResultKey(variantId: string, caseId: string): string {
422423
return `${variantId}::${caseId}`;
423424
}
424425

426+
export { evalHttpEnvGet, runWithEvalHttpEnv };
427+
425428
/**
426429
* Env vars the eval-only SSRF fixture exception in src/tools/ssrf-guard.ts
427430
* checks against. Shared by the agent process (must see EVAL_HTTP_URL so
@@ -433,23 +436,15 @@ export function httpFixtureEnv(fixture: { url: string; token: string }): Record<
433436
}
434437

435438
/**
436-
* Sets process.env vars for the duration of fn, restoring the prior values
437-
* (or deleting the key if it was unset) afterward, even on throw. The agent
438-
* runs in-process via runExec rather than as a spawned child, so fixture env
439-
* needed by in-process code (e.g. the eval-only SSRF exception) must be
440-
* applied to process.env directly instead of a child's env object.
439+
* Isolates `vars` for the duration of `fn` via async context (ALS), even when
440+
* sibling cells overlap under `--concurrency`. In-process readers (ssrf-guard)
441+
* see this cell's values through evalHttpEnvGet; one cell finishing cannot
442+
* delete a sibling's overlay. process.env is left alone so a restore cannot
443+
* clobber a concurrent cell. verify.sh still receives an explicit env object
444+
* at spawn (see scripts/eval-capability.ts).
441445
*/
442446
export async function withEnv<T>(vars: Record<string, string>, fn: () => Promise<T>): Promise<T> {
443-
const prior = new Map(Object.keys(vars).map((k) => [k, process.env[k]]));
444-
Object.assign(process.env, vars);
445-
try {
446-
return await fn();
447-
} finally {
448-
for (const [k, v] of prior) {
449-
if (v === undefined) delete process.env[k];
450-
else process.env[k] = v;
451-
}
452-
}
447+
return runWithEvalHttpEnv(vars, fn);
453448
}
454449

455450
/**

‎scripts/eval-capability.test.ts‎

Lines changed: 91 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,15 +1,33 @@
1-
import { afterEach, describe, expect, test } from "bun:test";
1+
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
22
import { mkdir, mkdtemp, writeFile, rm } from "node:fs/promises";
33
import { tmpdir } from "node:os";
44
import { join } from "node:path";
55
import { execFile } from "node:child_process";
66
import { promisify } from "node:util";
77

8-
import { initEvalGitRepo, parseArgs } from "./eval-capability.ts";
8+
import { initEvalGitRepo, mapPool, parseArgs } from "./eval-capability.ts";
99

1010
const execFileAsync = promisify(execFile);
1111

1212
describe("parseArgs", () => {
13+
const savedConcurrency = process.env.CORBITS_EVAL_CONCURRENCY;
14+
15+
const restoreConcurrency = (): void => {
16+
if (savedConcurrency === undefined) {
17+
delete process.env.CORBITS_EVAL_CONCURRENCY;
18+
} else {
19+
process.env.CORBITS_EVAL_CONCURRENCY = savedConcurrency;
20+
}
21+
};
22+
23+
afterEach(() => {
24+
restoreConcurrency();
25+
});
26+
27+
beforeEach(() => {
28+
delete process.env.CORBITS_EVAL_CONCURRENCY;
29+
});
30+
1331
test("--help does not require provider or model", () => {
1432
const opts = parseArgs(["--help"]);
1533
expect(opts.help).toBe(true);
@@ -61,6 +79,77 @@ describe("parseArgs", () => {
6179
expect(pair.provider).toBe("foo");
6280
expect(pair.model).toBe("bar");
6381
});
82+
83+
test("defaults concurrency to 1", () => {
84+
delete process.env.CORBITS_EVAL_CONCURRENCY;
85+
const opts = parseArgs(["--provider", "foo", "--model", "bar"]);
86+
expect(opts.concurrency).toBe(1);
87+
});
88+
89+
test("--concurrency 4 is accepted", () => {
90+
delete process.env.CORBITS_EVAL_CONCURRENCY;
91+
const opts = parseArgs(["--provider", "foo", "--model", "bar", "--concurrency", "4"]);
92+
expect(opts.concurrency).toBe(4);
93+
});
94+
95+
test("invalid --concurrency values throw", () => {
96+
const pair = ["--provider", "foo", "--model", "bar"] as const;
97+
expect(() => parseArgs([...pair, "--concurrency", "0"])).toThrow(/positive integer/);
98+
expect(() => parseArgs([...pair, "--concurrency", "-1"])).toThrow(/positive integer/);
99+
expect(() => parseArgs([...pair, "--concurrency", "1.5"])).toThrow(/positive integer/);
100+
expect(() => parseArgs([...pair, "--concurrency", "foo"])).toThrow(/positive integer/);
101+
});
102+
103+
test("CORBITS_EVAL_CONCURRENCY sets the default", () => {
104+
process.env.CORBITS_EVAL_CONCURRENCY = "3";
105+
const opts = parseArgs(["--provider", "foo", "--model", "bar"]);
106+
expect(opts.concurrency).toBe(3);
107+
});
108+
109+
test("--concurrency overrides CORBITS_EVAL_CONCURRENCY", () => {
110+
process.env.CORBITS_EVAL_CONCURRENCY = "8";
111+
const opts = parseArgs(["--provider", "foo", "--model", "bar", "--concurrency", "2"]);
112+
expect(opts.concurrency).toBe(2);
113+
});
114+
115+
test("invalid CORBITS_EVAL_CONCURRENCY throws", () => {
116+
process.env.CORBITS_EVAL_CONCURRENCY = "0";
117+
expect(() => parseArgs(["--provider", "foo", "--model", "bar"])).toThrow(
118+
/CORBITS_EVAL_CONCURRENCY must be a positive integer/,
119+
);
120+
});
121+
});
122+
123+
describe("mapPool", () => {
124+
test("N overlapping jobs with concurrency N finish in ~one job duration", async () => {
125+
const jobMs = 80;
126+
const n = 4;
127+
const start = Date.now();
128+
const results = await mapPool([0, 1, 2, 3], n, async (item) => {
129+
await new Promise((r) => setTimeout(r, jobMs));
130+
return item;
131+
});
132+
const elapsed = Date.now() - start;
133+
expect(results).toEqual([0, 1, 2, 3]);
134+
expect(elapsed).toBeLessThan(jobMs * 2);
135+
expect(elapsed).toBeGreaterThanOrEqual(jobMs - 20);
136+
});
137+
138+
test("preserves input order when later items finish first", async () => {
139+
const results = await mapPool([1, 2, 3], 3, async (item) => {
140+
await new Promise((r) => setTimeout(r, (4 - item) * 30));
141+
return item;
142+
});
143+
expect(results).toEqual([1, 2, 3]);
144+
});
145+
146+
test("empty input returns an empty array", async () => {
147+
expect(await mapPool([], 4, async (item) => item)).toEqual([]);
148+
});
149+
150+
test("rejects non-positive concurrency", async () => {
151+
await expect(mapPool([1], 0, async (item) => item)).rejects.toThrow(/positive integer/);
152+
});
64153
});
65154

66155
describe("initEvalGitRepo", () => {

‎scripts/eval-capability.ts‎

Lines changed: 55 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -68,6 +68,8 @@ type CliOptions = {
6868
verifyTimeoutMs: number;
6969
/** Runs per case×variant cell (gate runs use 5; freeze runs use 3). */
7070
repeats: number;
71+
/** Independent case×variant×repeat cells in parallel (default 1). */
72+
concurrency: number;
7173
dryRun: boolean;
7274
help: boolean;
7375
/**
@@ -93,18 +95,62 @@ function printUsage(): void {
9395
--agent-timeout-ms <n> Wall-clock limit for runExec (default 1200000)
9496
--verify-timeout-ms <n> Wall-clock limit for verify.sh (default 120000)
9597
--repeats <n> Runs per case×variant cell (default 1; gate runs use 5)
98+
--concurrency <n> Independent cells in parallel (default 1, env CORBITS_EVAL_CONCURRENCY)
9699
--dry-run List cases × variants only (still requires --provider/--model or --matrix)
97100
--allow-provider-fallback Allow resolved provider/model to differ from
98101
what was requested (default: hard-fail)
99102
-h, --help Show help
100103
`);
101104
}
102105

106+
function parsePositiveInteger(raw: string, label: string): number {
107+
const n = Number(raw);
108+
if (!Number.isInteger(n) || n <= 0) {
109+
throw new Error(`${label} must be a positive integer`);
110+
}
111+
return n;
112+
}
113+
114+
function defaultConcurrency(): number {
115+
const raw = process.env.CORBITS_EVAL_CONCURRENCY;
116+
if (raw === undefined || raw === "") return 1;
117+
return parsePositiveInteger(raw, "CORBITS_EVAL_CONCURRENCY");
118+
}
119+
120+
/**
121+
* Run `mapper` over `items` with at most `concurrency` in flight.
122+
* Results stay in input order even when later items finish first.
123+
*/
124+
export async function mapPool<T, R>(
125+
items: readonly T[],
126+
concurrency: number,
127+
mapper: (item: T, index: number) => Promise<R>,
128+
): Promise<R[]> {
129+
if (!Number.isInteger(concurrency) || concurrency <= 0) {
130+
throw new Error("concurrency must be a positive integer");
131+
}
132+
if (items.length === 0) return [];
133+
const results: R[] = new Array(items.length);
134+
let nextIndex = 0;
135+
const worker = async (): Promise<void> => {
136+
while (true) {
137+
const index = nextIndex;
138+
nextIndex += 1;
139+
if (index >= items.length) return;
140+
results[index] = await mapper(items[index]!, index);
141+
}
142+
};
143+
const workerCount = Math.min(concurrency, items.length);
144+
await Promise.all(Array.from({ length: workerCount }, () => worker()));
145+
return results;
146+
}
147+
103148
export function parseArgs(argv: readonly string[]): CliOptions {
104149
const opts: CliOptions = {
105150
caseSelector: "all",
106151
skipPermissions: true,
107152
repeats: 1,
153+
concurrency: defaultConcurrency(),
108154
dryRun: false,
109155
help: false,
110156
allowProviderFallback: false,
@@ -175,6 +221,9 @@ export function parseArgs(argv: readonly string[]): CliOptions {
175221
opts.repeats = n;
176222
break;
177223
}
224+
case "--concurrency":
225+
opts.concurrency = parsePositiveInteger(next(), "--concurrency");
226+
break;
178227
case "--dry-run":
179228
opts.dryRun = true;
180229
break;
@@ -771,6 +820,7 @@ async function main(): Promise<number> {
771820
}
772821

773822
console.log(`Repeats per cell: ${opts.repeats}`);
823+
console.log(`Concurrency: ${opts.concurrency}`);
774824

775825
if (opts.dryRun) {
776826
console.log("dry-run: no inference");
@@ -781,14 +831,15 @@ async function main(): Promise<number> {
781831
}
782832

783833
const startedAt = new Date().toISOString();
784-
const results: CaseResult[] = [];
785-
834+
const cells: Array<{ caseDef: EvalCase; variant: EvalVariant; repeat: number }> = [];
786835
for (const { caseDef, variant } of plan) {
787836
for (let repeat = 0; repeat < opts.repeats; repeat++) {
788-
const result = await runCase(caseDef, variant, opts, repeat);
789-
results.push(result);
837+
cells.push({ caseDef, variant, repeat });
790838
}
791839
}
840+
const results = await mapPool(cells, opts.concurrency, ({ caseDef, variant, repeat }) =>
841+
runCase(caseDef, variant, opts, repeat),
842+
);
792843

793844
const finishedAt = new Date().toISOString();
794845
const totals = summarizeRun(results);

0 commit comments

Comments
 (0)