Skip to content

Commit 84e5313

Browse files
committed
Refresh Codex instructions on exec boot and record eval diagnostics
Exec now awaits the same refreshCodexInstructions() the TUI uses, best-effort before first Codex inference, so capability/public SWE evals stop running on stale cached or bundled instructions. eval-capability.ts also stamps each result with a Codex instructions hash, the advertised tool list, and the requested reasoning effort for easier failure triage. Fixes CL-6693 https://linear.app/abklabs/issue/CL-6693
1 parent dba0d3f commit 84e5313

7 files changed

Lines changed: 250 additions & 2 deletions

File tree

‎evals/capability/lib.test.ts‎

Lines changed: 33 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -71,6 +71,7 @@ function sampleResult(over: Partial<CaseResult> = {}): CaseResult {
7171
repeat: over.repeat ?? 0,
7272
behaviors: over.behaviors ?? null,
7373
providerFallback: over.providerFallback ?? null,
74+
diagnostics: over.diagnostics ?? null,
7475
};
7576
}
7677

@@ -755,6 +756,38 @@ describe("parseEvalRunReport", () => {
755756
});
756757
});
757758

759+
test("round-trips diagnostics stamping on a case result", () => {
760+
const report = parseEvalRunReport({
761+
version: 3,
762+
provider: "xai",
763+
model: "grok",
764+
cases: [
765+
sampleResult({
766+
diagnostics: {
767+
codexInstructionsHash: "abc123def456",
768+
advertisedTools: ["read_file", "run_shell"],
769+
reasoningEffort: "high",
770+
},
771+
}),
772+
],
773+
});
774+
expect(report.cases[0]!.diagnostics).toEqual({
775+
codexInstructionsHash: "abc123def456",
776+
advertisedTools: ["read_file", "run_shell"],
777+
reasoningEffort: "high",
778+
});
779+
});
780+
781+
test("legacy reports with no diagnostics parse to null", () => {
782+
const report = parseEvalRunReport({
783+
version: 3,
784+
provider: "xai",
785+
model: "grok",
786+
cases: [sampleResult()],
787+
});
788+
expect(report.cases[0]!.diagnostics).toBeNull();
789+
});
790+
758791
test("legacy reports default repeat 0 and null behaviors", () => {
759792
const report = parseEvalRunReport({
760793
version: 2,

‎evals/capability/lib.ts‎

Lines changed: 23 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -134,9 +134,19 @@ export type CaseResult = {
134134
behaviors: BehaviorMetrics | null;
135135
/** Set when the resolved provider/model differed from what was requested. */
136136
providerFallback: ProviderFallbackInfo | null;
137+
/** Per-cell diagnostics for debugging eval failures; null when unavailable. */
138+
diagnostics: EvalDiagnostics | null;
137139
textPreview?: string;
138140
};
139141

142+
export type EvalDiagnostics = {
143+
/** Short identity (hash) of the pinned Codex instructions text in use; null for non-Codex providers. */
144+
codexInstructionsHash: string | null;
145+
/** Built-in tool names advertised to the model for this run. */
146+
advertisedTools: readonly string[];
147+
reasoningEffort: string | null;
148+
};
149+
140150
export type MetricStats = {
141151
min: number;
142152
median: number;
@@ -726,10 +736,23 @@ function parseCaseResult(raw: unknown): CaseResult {
726736
: 0,
727737
behaviors: parseBehaviorMetrics(raw.behaviors),
728738
providerFallback: parseProviderFallback(raw.providerFallback),
739+
diagnostics: parseEvalDiagnostics(raw.diagnostics),
729740
...(typeof raw.textPreview === "string" ? { textPreview: raw.textPreview } : {}),
730741
};
731742
}
732743

744+
function parseEvalDiagnostics(raw: unknown): EvalDiagnostics | null {
745+
if (!isRecord(raw)) return null;
746+
if (!Array.isArray(raw.advertisedTools)) return null;
747+
const advertisedTools = raw.advertisedTools.filter((t): t is string => typeof t === "string");
748+
return {
749+
codexInstructionsHash:
750+
typeof raw.codexInstructionsHash === "string" ? raw.codexInstructionsHash : null,
751+
advertisedTools,
752+
reasoningEffort: typeof raw.reasoningEffort === "string" ? raw.reasoningEffort : null,
753+
};
754+
}
755+
733756
function parseProviderFallback(raw: unknown): ProviderFallbackInfo | null {
734757
if (!isRecord(raw)) return null;
735758
const resolvedProvider = raw.resolvedProvider;

‎scripts/eval-capability.test.ts‎

Lines changed: 43 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -5,10 +5,32 @@ import { join } from "node:path";
55
import { execFile } from "node:child_process";
66
import { promisify } from "node:util";
77

8-
import { initEvalGitRepo, mapPool, parseArgs } from "./eval-capability.ts";
8+
import { initEvalGitRepo, mapPool, parseArgs, buildEvalDiagnostics } from "./eval-capability.ts";
9+
import type { Config } from "../src/config/index.ts";
910

1011
const execFileAsync = promisify(execFile);
1112

13+
function sampleConfig(over: Partial<Config> = {}): Config {
14+
return {
15+
configured: true,
16+
apiKey: "key",
17+
baseURL: "https://example.test",
18+
model: "gpt-5",
19+
providerName: "openai",
20+
cwd: process.cwd(),
21+
task: "do it",
22+
force: true,
23+
dangerouslySkipPermissions: true,
24+
skipPermissionsFromSettings: false,
25+
auto: false,
26+
command: "exec",
27+
globalSettingsPath: "/dev/null",
28+
providers: [],
29+
sessionId: "sess-1",
30+
...over,
31+
} as Config;
32+
}
33+
1234
describe("parseArgs", () => {
1335
const savedConcurrency = process.env.CORBITS_EVAL_CONCURRENCY;
1436

@@ -232,3 +254,23 @@ describe("initEvalGitRepo", () => {
232254
}
233255
});
234256
});
257+
258+
describe("buildEvalDiagnostics", () => {
259+
test("non-Codex provider gets a null instructions hash and the advertised tool list", () => {
260+
const diagnostics = buildEvalDiagnostics(sampleConfig({ providerName: "openai" }));
261+
expect(diagnostics.codexInstructionsHash).toBeNull();
262+
expect(diagnostics.advertisedTools).toContain("read_file");
263+
expect(diagnostics.advertisedTools).toContain("run_shell");
264+
expect(diagnostics.reasoningEffort).toBeNull();
265+
});
266+
267+
test("Codex provider gets a non-null instructions hash", () => {
268+
const diagnostics = buildEvalDiagnostics(sampleConfig({ providerName: "codex/default" }));
269+
expect(diagnostics.codexInstructionsHash).toMatch(/^[0-9a-f]{12}$/);
270+
});
271+
272+
test("echoes back the configured reasoning effort", () => {
273+
const diagnostics = buildEvalDiagnostics(sampleConfig({ reasoningEffort: "high" }));
274+
expect(diagnostics.reasoningEffort).toBe("high");
275+
});
276+
});

‎scripts/eval-capability.ts‎

Lines changed: 27 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -14,9 +14,13 @@ import { tmpdir } from "node:os";
1414
import { join, dirname, resolve } from "node:path";
1515
import { fileURLToPath } from "node:url";
1616
import { spawn } from "node:child_process";
17-
import { loadConfig } from "../src/config/index.js";
17+
import { loadConfig, type Config } from "../src/config/index.js";
1818
import { runExec } from "../src/exec/runner.js";
1919
import { SETTINGS_DIR_NAME } from "../src/branding.js";
20+
import { codexProfileFromProviderName } from "../src/config/codex-providers.js";
21+
import { codexInstructionsHash } from "../src/auth/codex/instructions.js";
22+
import { advertisedToolNamesForSessionMode } from "../src/agent/tool-search.js";
23+
import { detectLanguageServerAvailable } from "../src/agent/lsp-availability.js";
2024
import {
2125
loadEvalCases,
2226
filterCases,
@@ -41,6 +45,7 @@ import {
4145
type EvalVariant,
4246
type EvalTokenUsage,
4347
type ProviderFallbackInfo,
48+
type EvalDiagnostics,
4449
} from "../evals/capability/lib.js";
4550
import {
4651
deriveBehaviorMetrics,
@@ -531,6 +536,24 @@ async function resolveVariantLabels(
531536
};
532537
}
533538

539+
/**
540+
* Per-cell diagnostics for debugging eval failures: which Codex instructions
541+
* text was pinned, which built-in tools the model was offered, and the
542+
* requested reasoning effort. Mirrors the exec runner's own session-mode/tool
543+
* gating (see src/agent/tool-search.ts) rather than re-running toolset setup.
544+
*/
545+
export function buildEvalDiagnostics(config: Config): EvalDiagnostics {
546+
const codexProfile = codexProfileFromProviderName(config.providerName);
547+
const advertisedTools = advertisedToolNamesForSessionMode("orchestrator", {
548+
languageServerAvailable: detectLanguageServerAvailable(config.cwd),
549+
});
550+
return {
551+
codexInstructionsHash: codexProfile !== undefined ? codexInstructionsHash() : null,
552+
advertisedTools,
553+
reasoningEffort: config.reasoningEffort ?? null,
554+
};
555+
}
556+
534557
function failResult(
535558
caseDef: EvalCase,
536559
variant: EvalVariant,
@@ -568,6 +591,7 @@ function failResult(
568591
repeat,
569592
behaviors: null,
570593
providerFallback: null,
594+
diagnostics: null,
571595
...partial,
572596
};
573597
}
@@ -633,6 +657,7 @@ async function runCase(
633657
);
634658
}
635659

660+
const diagnostics = buildEvalDiagnostics(config);
636661
const agentStarted = Date.now();
637662
// runExec runs the agent in-process (no child, unlike verify.sh below), so
638663
// the fixture origin must reach it via process.env directly for the
@@ -770,6 +795,7 @@ async function runCase(
770795
repeat,
771796
behaviors,
772797
providerFallback,
798+
diagnostics,
773799
textPreview,
774800
};
775801
} catch (err) {
Lines changed: 102 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,102 @@
1+
import { afterEach, beforeEach, describe, expect, mock, test } from "bun:test";
2+
3+
// Reused by both the TUI and exec boot paths (src/tui/runner.ts,
4+
// src/exec/runner.ts) to refresh the pinned Codex instructions before first
5+
// Codex inference. This exercises the shared refresh/fallback logic directly,
6+
// with disk I/O faked so tests never touch the real ~/.corbits cache.
7+
8+
let fakeDisk = new Map<string, string>();
9+
10+
mock.module("node:fs", () => ({
11+
readFileSync: (path: string) => {
12+
const contents = fakeDisk.get(path);
13+
if (contents === undefined) {
14+
const err = new Error("ENOENT") as NodeJS.ErrnoException;
15+
err.code = "ENOENT";
16+
throw err;
17+
}
18+
return contents;
19+
},
20+
writeFileSync: (path: string, contents: string) => {
21+
fakeDisk.set(path, contents);
22+
},
23+
mkdirSync: () => undefined,
24+
}));
25+
26+
const { refreshCodexInstructions, codexInstructions, codexInstructionsHash } = await import(
27+
"./instructions.js"
28+
);
29+
const { GPT_5_CODEX_PROMPT } = await import("./prompts/gpt-5-codex.js");
30+
31+
const VALID_PROMPT = `You are Codex${"x".repeat(1200)}`;
32+
33+
function mockFetchSequence(tag: string, promptResponse: () => Response): typeof fetch {
34+
return (async (input: RequestInfo | URL) => {
35+
const url = String(input);
36+
if (url.includes("releases/latest")) {
37+
return new Response(JSON.stringify({ tag_name: tag }), { status: 200 });
38+
}
39+
return promptResponse();
40+
}) as typeof fetch;
41+
}
42+
43+
describe("refreshCodexInstructions", () => {
44+
const originalFetch = global.fetch;
45+
46+
beforeEach(() => {
47+
fakeDisk = new Map();
48+
});
49+
50+
afterEach(() => {
51+
global.fetch = originalFetch;
52+
});
53+
54+
test("bundled copy is used before any refresh", () => {
55+
expect(codexInstructions()).toBe(GPT_5_CODEX_PROMPT);
56+
});
57+
58+
test("updates in-memory instructions on a successful fetch", async () => {
59+
global.fetch = mockFetchSequence("rust-v1.2.3", () => new Response(VALID_PROMPT, { status: 200 }));
60+
61+
await refreshCodexInstructions();
62+
expect(codexInstructions()).toBe(VALID_PROMPT);
63+
});
64+
65+
test("falls back without throwing the run when the release lookup network call fails", async () => {
66+
const before = codexInstructions();
67+
global.fetch = (() => Promise.reject(new Error("network down"))) as unknown as typeof fetch;
68+
69+
// The function itself rejects; callers (TUI/exec boot) catch this and
70+
// keep running on cache/bundled instructions — see src/exec/runner.ts and
71+
// src/tui/runner.ts refresh call sites.
72+
await expect(refreshCodexInstructions()).rejects.toThrow("network down");
73+
expect(codexInstructions()).toBe(before);
74+
});
75+
76+
test("falls back without throwing when the prompt fetch returns a non-200", async () => {
77+
const before = codexInstructions();
78+
global.fetch = mockFetchSequence("rust-v1.2.3", () => new Response("not found", { status: 404 }));
79+
80+
await expect(refreshCodexInstructions()).rejects.toThrow(/HTTP 404/);
81+
expect(codexInstructions()).toBe(before);
82+
});
83+
84+
test("rejects a 200 response whose body is not a valid Codex prompt (CDN error page)", async () => {
85+
const before = codexInstructions();
86+
global.fetch = mockFetchSequence("rust-v1.2.3", () => new Response("<html>oops</html>", { status: 200 }));
87+
88+
await expect(refreshCodexInstructions()).rejects.toThrow(/unexpected body/);
89+
expect(codexInstructions()).toBe(before);
90+
});
91+
92+
test("codexInstructionsHash reflects the currently resolved instructions text", async () => {
93+
const hashBefore = codexInstructionsHash();
94+
expect(hashBefore).toMatch(/^[0-9a-f]{12}$/);
95+
96+
const otherPrompt = `You are Codex${"y".repeat(1200)}`;
97+
global.fetch = mockFetchSequence("rust-v1.2.4", () => new Response(otherPrompt, { status: 200 }));
98+
await refreshCodexInstructions();
99+
100+
expect(codexInstructionsHash()).not.toBe(hashBefore);
101+
});
102+
});

‎src/auth/codex/instructions.ts‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
11
import { readFileSync, writeFileSync, mkdirSync } from "node:fs";
2+
import { createHash } from "node:crypto";
23
import { homedir } from "node:os";
34
import { join } from "node:path";
45
import { GPT_5_CODEX_PROMPT } from "./prompts/gpt-5-codex.js";
@@ -42,6 +43,14 @@ export function codexInstructions(): string {
4243
return instructions;
4344
}
4445

46+
/**
47+
* Short identity for the in-use instructions text, for eval/diagnostic
48+
* records — never sent to the model.
49+
*/
50+
export function codexInstructionsHash(): string {
51+
return createHash("sha256").update(codexInstructions()).digest("hex").slice(0, 12);
52+
}
53+
4554
async function latestReleaseTag(): Promise<string> {
4655
const res = await fetch(RELEASES_LATEST, { headers: { accept: "application/vnd.github+json" } });
4756
if (!res.ok) throw new Error(`Codex release lookup failed (HTTP ${String(res.status)}).`);

‎src/exec/runner.ts‎

Lines changed: 13 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -32,6 +32,7 @@ import { DIRECTOR_REGISTRY } from "../agent/directors/registry.js";
3232
import type { DirectorId } from "../agent/directors/types.js";
3333
import { createInferenceDependencies } from "../provider/inference-dependencies.js";
3434
import { getValidCodexToken } from "../auth/codex/session.js";
35+
import { refreshCodexInstructions } from "../auth/codex/instructions.js";
3536
import { getValidXaiToken } from "../auth/xai/session.js";
3637
import { seedPricingMetadataFromCache } from "../cost/pricing-metadata.js";
3738
import { defaultPricingCachePath } from "../cost/pricing-fetcher.js";
@@ -572,6 +573,18 @@ export async function runExec(config: Config): Promise<ExecResult> {
572573
liveSources[0] ??
573574
buildInitialSourceFallback();
574575

576+
// Refresh pinned Codex instructions before first inference, same as the
577+
// TUI path. Best-effort: a network failure falls back to the disk cache
578+
// or bundled copy without failing the run. Exec is one-shot (no long-lived
579+
// session to catch up later), so this is awaited rather than fire-and-forget.
580+
if (initialCodexProfile !== undefined) {
581+
await refreshCodexInstructions().catch((err: unknown) => {
582+
logger.warn("Codex instructions refresh failed: {error}", {
583+
error: formatCaughtError(err),
584+
});
585+
});
586+
}
587+
575588
// Refresh OAuth tokens before first inference when starting on codex/xai.
576589
if (initialCodexProfile !== undefined) {
577590
const { access } = await getValidCodexToken(initialCodexProfile);

0 commit comments

Comments
 (0)