diff --git a/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts b/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts index 9d1c4aa12..f1db77304 100644 --- a/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts +++ b/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts @@ -1,12 +1,13 @@ -// Tests for the CL-6204 context-budget resolver: reads a per-model -// `numCtx` hint off an `InferenceSource.quirks` bag shaped like -// `@corbits/ollama-adapter`'s `OllamaAdapterConfig`, falling back to a -// conservative default for any other shape. +// Tests for the context-budget resolver: quirks `numCtx` pin, then the +// advertised catalog window for the model name, never a baked 8k-token +// (32k-char) cap for frontier models. import { expect, test } from "bun:test"; import { + advertisedContextWindowTokens, readNumCtxHint, resolveContextBudgetChars, + resolveContextWindowTokens, resolveHardContextLimitChars, } from "./context-budget"; @@ -27,6 +28,10 @@ test("readNumCtxHint: unrecognized or absent quirks resolve to undefined", () => expect(readNumCtxHint("not-an-object", "claude")).toBeUndefined(); }); +test("readNumCtxHint: a top-level numCtx pin is visible without the Ollama bag shape", () => { + expect(readNumCtxHint({ numCtx: 200_000 }, "claude-sonnet-5")).toBe(200_000); +}); + test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () => { const small = resolveContextBudgetChars( { default: { numCtx: 32_000 } }, @@ -40,11 +45,48 @@ test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () => expect(large).toBeGreaterThan(small); }); -test("resolveContextBudgetChars: unknown model falls back to the conservative default", () => { +test("resolveContextWindowTokens: quirks pin wins over the advertised catalog window", () => { + expect( + resolveContextWindowTokens( + { default: { numCtx: 8_192 } }, + "claude-sonnet-5", + ), + ).toBe(8_192); +}); + +test("resolveContextWindowTokens: a frontier catalog model gets its advertised window, not a 32k-char cap", () => { + const claudeTokens = resolveContextWindowTokens(undefined, "claude-sonnet-5"); + const claudeHard = resolveHardContextLimitChars(undefined, "claude-sonnet-5"); + const oldDefaultHardChars = 8_000 * 4; + + expect(claudeTokens).toBe(200_000); + expect(claudeHard).toBeGreaterThan(oldDefaultHardChars); + expect(claudeHard).toBe(200_000 * 4); +}); + +test("resolveContextWindowTokens: an Ollama catalog model gets its native window without quirks", () => { + expect(resolveContextWindowTokens(undefined, "gpt-oss:20b")).toBe(131_072); + expect(resolveContextWindowTokens(undefined, "qwen3.8:27b")).toBe(32_768); + expect( + resolveHardContextLimitChars(undefined, "gpt-oss:20b"), + ).toBeGreaterThan(resolveHardContextLimitChars(undefined, "qwen3.8:27b")); +}); + +test("advertisedContextWindowTokens: a relay-prefixed name still matches the catalog model", () => { + expect(advertisedContextWindowTokens("anthropic/claude-sonnet-5")).toBe( + 200_000, + ); + expect(advertisedContextWindowTokens("openai/gpt-4.1")).toBe(1_047_576); +}); + +test("resolveContextBudgetChars: unknown model falls back to a hosted-sized window, not 8k tokens", () => { const withoutQuirks = resolveContextBudgetChars(undefined, "unknown-model"); + const hard = resolveHardContextLimitChars(undefined, "unknown-model"); expect(withoutQuirks).toBeGreaterThan(0); expect(Number.isFinite(withoutQuirks)).toBe(true); + expect(hard).toBeGreaterThan(8_000 * 4); + expect(hard).toBe(128_000 * 4); }); test("resolveHardContextLimitChars: sits above the headroomed budget for the same source", () => { diff --git a/apps/sidecar/src/workflow-substrate-factory/context-budget.ts b/apps/sidecar/src/workflow-substrate-factory/context-budget.ts index bfad27b0c..b7b44b91d 100644 --- a/apps/sidecar/src/workflow-substrate-factory/context-budget.ts +++ b/apps/sidecar/src/workflow-substrate-factory/context-budget.ts @@ -1,40 +1,119 @@ -// Best-effort per-model context-window budget for the workbench -// director's compaction gate (CL-6204). +// Per-model context-window budget for the workbench director's +// compaction gate. // -// Reads the same `numCtx` override shape `@corbits/ollama-adapter` -// resolves against a live request (`OllamaAdapterConfig`'s -// `default`/`perModel` bag) directly off `InferenceSource.quirks`, so a -// deployment that already pins a per-model `num_ctx` for Ollama gets -// that same number as its compaction budget instead of a second, -// drifting constant. `quirks` is `unknown` at this boundary (an -// operator-authored, per-source passthrough bag), so every read here is -// defensive rather than a validating parse: an unrecognized shape (every -// non-Ollama source today) falls back to a conservative default sized -// for the smallest models this repo deploys against, rather than -// throwing or silently trusting a shape the adapter itself doesn't own -// at this call site. +// Window size is the resolved model's advertised native context, not a +// baked 8k-token (32k-char) cap: // -// Token counts are estimated from character counts (`CHARS_PER_TOKEN_ESTIMATE`) -// -- no tokenizer is available at this layer -- so the budget is -// deliberately conservative (see `COMPACTION_HEADROOM`), leaving room -// for the system prompt, tool definitions, and the model's own reply -// inside the same window. +// 1. `InferenceSource.quirks` -- the same `numCtx` override shape +// `@corbits/ollama-adapter` resolves (`OllamaAdapterConfig`'s +// `default`/`perModel` bag), plus a few other common token-window +// field names operators actually pin. +// 2. The pinned catalog's advertised native window for that model +// name (Ollama table entries and hosted-family prefixes). +// 3. A modern hosted-model fallback (128k tokens), used only when +// neither quirks nor the catalog name the model. Compact (see +// `COMPACTION_HEADROOM`) remains the safety net inside that +// window; `CONTEXT_OVERFLOW_MESSAGE` is reserved for the hard +// edge. +// +// `quirks` is `unknown` at this boundary (an operator-authored, +// per-source passthrough bag), so every read here is defensive rather +// than a validating parse. +// +// Token counts are estimated from character counts +// (`CHARS_PER_TOKEN_ESTIMATE`) -- no tokenizer is available at this +// layer -- so the budget is deliberately conservative (see +// `COMPACTION_HEADROOM`), leaving room for the system prompt, tool +// definitions, and the model's own reply inside the same window. -const DEFAULT_CONTEXT_BUDGET_TOKENS = 8_000; +const FALLBACK_CONTEXT_WINDOW_TOKENS = 128_000; const CHARS_PER_TOKEN_ESTIMATE = 4; const COMPACTION_HEADROOM = 0.6; -function readNumCtxFromBag(bag: Record): number | undefined { - const numCtx = bag.numCtx; - return typeof numCtx === "number" && numCtx > 0 ? numCtx : undefined; +// Exact names from `@corbits/inference-catalog`'s Ollama native-window +// table, then longest-prefix families for hosted catalog models. +// Longest prefix wins so `gpt-4.1` is not swallowed by `gpt-4`. +const ADVERTISED_CONTEXT_WINDOWS: readonly (readonly [string, number])[] = [ + ["qwen3.5:9b-mlx", 32_768], + ["qwen3.8:27b", 32_768], + ["llama3.1:8b", 131_072], + ["gpt-oss:20b", 131_072], + ["gpt-4-turbo", 128_000], + ["gpt-4.1", 1_047_576], + ["llama3.1", 131_072], + ["gpt-oss", 131_072], + ["gpt-4o", 128_000], + ["claude", 200_000], + ["gemini", 1_048_576], + ["gpt-5", 400_000], + ["gpt-4", 8_192], + ["grok", 256_000], + ["qwen3", 32_768], + ["llama", 131_072], + ["kimi", 128_000], + ["deepseek", 128_000], + ["minimax", 128_000], + ["glm", 128_000], + ["o1", 200_000], + ["o3", 200_000], + ["o4", 200_000], +]; + +function positiveTokenCount(value: unknown): number | undefined { + return typeof value === "number" && Number.isFinite(value) && value > 0 + ? value + : undefined; +} + +function readWindowFromBag(bag: Record): number | undefined { + return ( + positiveTokenCount(bag.numCtx) ?? + positiveTokenCount(bag.contextWindow) ?? + positiveTokenCount(bag.contextLength) ?? + positiveTokenCount(bag.maxContextTokens) + ); +} + +function catalogModelKey(model: string): string { + const trimmed = model.trim().toLowerCase(); + const slash = trimmed.lastIndexOf("/"); + return slash === -1 ? trimmed : trimmed.slice(slash + 1); +} + +/** + * Advertised native context window in tokens for a catalog model name. + * Exact Ollama table entries win; otherwise the longest matching hosted + * family prefix. `undefined` if this module has no published window for + * the name. + */ +export function advertisedContextWindowTokens( + model: string, +): number | undefined { + const key = catalogModelKey(model); + if (key.length === 0) { + return undefined; + } + let best: { prefixLength: number; tokens: number } | undefined; + for (const [prefix, tokens] of ADVERTISED_CONTEXT_WINDOWS) { + if (key === prefix) { + return tokens; + } + if ( + key.startsWith(prefix) && + (best === undefined || prefix.length > best.prefixLength) + ) { + best = { prefixLength: prefix.length, tokens }; + } + } + return best?.tokens; } /** - * Best-effort read of a per-model `numCtx` off an `InferenceSource.quirks` - * bag shaped like `@corbits/ollama-adapter`'s `OllamaAdapterConfig` - * (`{ default?: { numCtx? }, perModel?: { [model]: { numCtx? } } }`). - * Returns `undefined` for any other shape rather than throwing -- - * `quirks` is provider-specific and most sources carry none of this. + * Best-effort read of a per-model window hint off an + * `InferenceSource.quirks` bag. Understands `@corbits/ollama-adapter`'s + * `OllamaAdapterConfig` (`{ default?: { numCtx? }, perModel?: { [model]: + * { numCtx? } } }`) and a few other common token-window field names. + * Returns `undefined` for any unrecognized shape rather than throwing. */ export function readNumCtxHint( quirks: unknown, @@ -48,7 +127,7 @@ export function readNumCtxHint( if (typeof perModel === "object" && perModel !== null) { const entry = (perModel as Record)[model]; if (typeof entry === "object" && entry !== null) { - const fromPerModel = readNumCtxFromBag(entry as Record); + const fromPerModel = readWindowFromBag(entry as Record); if (fromPerModel !== undefined) { return fromPerModel; } @@ -56,23 +135,48 @@ export function readNumCtxHint( } const base = bag.default; if (typeof base === "object" && base !== null) { - return readNumCtxFromBag(base as Record); + const fromDefault = readWindowFromBag(base as Record); + if (fromDefault !== undefined) { + return fromDefault; + } } - return undefined; + return readWindowFromBag(bag); +} + +/** + * Resolved context window in tokens for one `InferenceSource`: quirks + * pin, then the advertised catalog window, then the hosted-model + * fallback. Compact headroom is applied by the char-budget helpers, + * not here. + */ +export function resolveContextWindowTokens( + quirks: unknown, + model: string, +): number { + return ( + readNumCtxHint(quirks, model) ?? + advertisedContextWindowTokens(model) ?? + FALLBACK_CONTEXT_WINDOW_TOKENS + ); +} + +function toChars(tokens: number, headroom: number): number { + return Math.floor(tokens * CHARS_PER_TOKEN_ESTIMATE * headroom); } /** * Resolve the compaction budget, in characters, for one `InferenceSource`. - * Applies `COMPACTION_HEADROOM` on top of the resolved (or default) - * `numCtx` so compaction fires with room left in the window, not at its - * hard edge. + * Applies `COMPACTION_HEADROOM` on top of the resolved window so + * compaction fires with room left, not at the hard edge. */ export function resolveContextBudgetChars( quirks: unknown, model: string, ): number { - const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS; - return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE * COMPACTION_HEADROOM); + return toChars( + resolveContextWindowTokens(quirks, model), + COMPACTION_HEADROOM, + ); } /** @@ -85,6 +189,5 @@ export function resolveHardContextLimitChars( quirks: unknown, model: string, ): number { - const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS; - return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE); + return toChars(resolveContextWindowTokens(quirks, model), 1); } diff --git a/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts b/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts index 80104b171..c33d0ffaa 100644 --- a/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts +++ b/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts @@ -248,6 +248,32 @@ test("context budget: a short conversation under budget is untouched (infers nor expect(typesOf(actions)).toEqual(["infer"]); }); +test("context budget: message.received over budget but under the hard limit still infers", async () => { + const director = createWorkbenchDirector( + "you are a test agent", + [], + {}, + { + budgetChars: 100, + hardLimitChars: 1_000_000, + compactorName: "summarize-budgeted-turns", + }, + ); + const overBudget = stateWithTurns([ + conversationTurn("a".repeat(200)), + conversationTurn("b".repeat(200)), + ]); + + const actions = await director.decide( + { type: "message.received", message: { id: "m1", content: "hi" } as never }, + overBudget, + caps, + ); + + expect(typesOf(actions)).toEqual(["infer"]); + expect(typesOf(actions)).not.toContain("compact"); +}); + test("context budget: tool-heavy history past the hard limit is caught even though every turn's text excerpt is short", async () => { // The reviewer's exact repro: 10 turns each carrying a 20,000-char // tool_result. Measured by placeholder length this was ~160 chars diff --git a/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts b/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts index e8362223c..9bde5f143 100644 --- a/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts +++ b/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts @@ -1,5 +1,5 @@ // Workbench director: DefaultDirector plus an empty-turn retry and a -// context-budget gate (CL-6204). +// context-budget gate. // // `@intx/inference`'s DefaultDirector checkpoints and waits when // inference.done has no text and no tool calls. The human then sits in a @@ -11,31 +11,31 @@ // We compose that path rather than reimplement it. // // Context budget (`contextBudget`, optional -- absent means unbudgeted, -// today's pre-CL-6204 behavior): before letting any inner decision that -// includes `infer` through, checks the turn history against the +// today's pre-CL-6204 behavior): checks the turn history against the // model's real context window (`resolveContextBudgetChars` / // `resolveHardContextLimitChars` in `./context-budget`, sized from -// `InferenceSource.quirks`). +// `InferenceSource.quirks` and the advertised catalog window). // // - Over the hard limit (no headroom left at all): this is Ollama's // silent-truncation case made honest -- reply with the same // "exceeded the model's context limit" message hosted providers // produce via `inference.error`'s `context_overflow` category // (`@intx/inference`'s `default-director.js`), instead of sending a -// request that would truncate server-side with no error. +// request that would truncate server-side with no error. Compact +// cannot help here, so `CONTEXT_OVERFLOW_MESSAGE` is the only reply. // - Over the (headroomed) budget but under the hard limit, on a // non-final `tool.done` in a multi-call batch (DefaultDirector -// returns `[]` for every `tool.done` before the batch's last): -// fire `caps.compact` here instead of the no-op. This is the one -// point in the reactor's event flow where `compact` cannot stall a -// reply -- the batch's remaining `tool.done` events are already -// enqueued (`@intx/inference`'s `reactor.js` `executeTools` enqueues -// every result from one `Promise.all` before the reactor dequeues -// any of them) and will still drive the eventual re-infer. Firing -// compact on `message.received` or the batch's *last* `tool.done` -// would instead stall that message's reply waiting for a follow-up -// event the reactor never produces (`compactors.ts`'s header -// comment) -- this director never does that. +// returns `[]` for every `tool.done` before the batch's last): fire +// `caps.compact` instead of the no-op. The batch's remaining +// `tool.done` events are already enqueued (`@intx/inference`'s +// `reactor.js` `executeTools` enqueues every result from one +// `Promise.all` before the reactor dequeues any of them) and will +// still drive the eventual re-infer. Do not compact as the sole +// terminal action on `message.received`: the reactor forbids +// compact+infer in one cycle and does not re-enter the director +// after compact, so compact-only would leave the inbound message +// unanswered. Over-budget inbound still infers; compaction waits +// for that next safe point. // - Otherwise: let the inner decision through unchanged. Compaction is // deferred to the next safe point rather than forced here. // @@ -218,13 +218,19 @@ export class WorkbenchDirector implements ReactorDirector { const list = Array.isArray(actions) ? actions : [actions]; const chars = estimateTurnsChars(state.turns); - if (list.some((action) => action.type === "infer")) { - if (chars > this.contextBudget.hardLimitChars) { + if (chars > this.contextBudget.hardLimitChars) { + if ( + list.some((action) => action.type === "infer") || + event.type === "message.received" + ) { return [ capabilities.checkpoint("context-overflow"), capabilities.reply(CONTEXT_OVERFLOW_MESSAGE), ]; } + } + + if (list.some((action) => action.type === "infer")) { return undefined; }