From a864d3391b0ef1a0ee10d896ccbc2b615d205c18 Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Fri, 28 Aug 2026 08:43:20 -0700 Subject: [PATCH 1/2] Size sidecar context budget from the model's real window Compaction used an 8k-token default (32k characters) whenever quirks did not pin numCtx, so frontier hosted models overflowed far inside their advertised windows. Resolve the window from quirks, then the catalog's advertised native size, and compact on message.received when over the headroomed budget but still under the hard limit. --- .../context-budget.test.ts | 52 ++++- .../context-budget.ts | 179 ++++++++++++++---- .../workbench-director.test.ts | 26 +++ .../workbench-director.ts | 53 ++++-- 4 files changed, 249 insertions(+), 61 deletions(-) diff --git a/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts b/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts index 9d1c4aa12..f1db77304 100644 --- a/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts +++ b/apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts @@ -1,12 +1,13 @@ -// Tests for the CL-6204 context-budget resolver: reads a per-model -// `numCtx` hint off an `InferenceSource.quirks` bag shaped like -// `@corbits/ollama-adapter`'s `OllamaAdapterConfig`, falling back to a -// conservative default for any other shape. +// Tests for the context-budget resolver: quirks `numCtx` pin, then the +// advertised catalog window for the model name, never a baked 8k-token +// (32k-char) cap for frontier models. import { expect, test } from "bun:test"; import { + advertisedContextWindowTokens, readNumCtxHint, resolveContextBudgetChars, + resolveContextWindowTokens, resolveHardContextLimitChars, } from "./context-budget"; @@ -27,6 +28,10 @@ test("readNumCtxHint: unrecognized or absent quirks resolve to undefined", () => expect(readNumCtxHint("not-an-object", "claude")).toBeUndefined(); }); +test("readNumCtxHint: a top-level numCtx pin is visible without the Ollama bag shape", () => { + expect(readNumCtxHint({ numCtx: 200_000 }, "claude-sonnet-5")).toBe(200_000); +}); + test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () => { const small = resolveContextBudgetChars( { default: { numCtx: 32_000 } }, @@ -40,11 +45,48 @@ test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () => expect(large).toBeGreaterThan(small); }); -test("resolveContextBudgetChars: unknown model falls back to the conservative default", () => { +test("resolveContextWindowTokens: quirks pin wins over the advertised catalog window", () => { + expect( + resolveContextWindowTokens( + { default: { numCtx: 8_192 } }, + "claude-sonnet-5", + ), + ).toBe(8_192); +}); + +test("resolveContextWindowTokens: a frontier catalog model gets its advertised window, not a 32k-char cap", () => { + const claudeTokens = resolveContextWindowTokens(undefined, "claude-sonnet-5"); + const claudeHard = resolveHardContextLimitChars(undefined, "claude-sonnet-5"); + const oldDefaultHardChars = 8_000 * 4; + + expect(claudeTokens).toBe(200_000); + expect(claudeHard).toBeGreaterThan(oldDefaultHardChars); + expect(claudeHard).toBe(200_000 * 4); +}); + +test("resolveContextWindowTokens: an Ollama catalog model gets its native window without quirks", () => { + expect(resolveContextWindowTokens(undefined, "gpt-oss:20b")).toBe(131_072); + expect(resolveContextWindowTokens(undefined, "qwen3.8:27b")).toBe(32_768); + expect( + resolveHardContextLimitChars(undefined, "gpt-oss:20b"), + ).toBeGreaterThan(resolveHardContextLimitChars(undefined, "qwen3.8:27b")); +}); + +test("advertisedContextWindowTokens: a relay-prefixed name still matches the catalog model", () => { + expect(advertisedContextWindowTokens("anthropic/claude-sonnet-5")).toBe( + 200_000, + ); + expect(advertisedContextWindowTokens("openai/gpt-4.1")).toBe(1_047_576); +}); + +test("resolveContextBudgetChars: unknown model falls back to a hosted-sized window, not 8k tokens", () => { const withoutQuirks = resolveContextBudgetChars(undefined, "unknown-model"); + const hard = resolveHardContextLimitChars(undefined, "unknown-model"); expect(withoutQuirks).toBeGreaterThan(0); expect(Number.isFinite(withoutQuirks)).toBe(true); + expect(hard).toBeGreaterThan(8_000 * 4); + expect(hard).toBe(128_000 * 4); }); test("resolveHardContextLimitChars: sits above the headroomed budget for the same source", () => { diff --git a/apps/sidecar/src/workflow-substrate-factory/context-budget.ts b/apps/sidecar/src/workflow-substrate-factory/context-budget.ts index bfad27b0c..b7b44b91d 100644 --- a/apps/sidecar/src/workflow-substrate-factory/context-budget.ts +++ b/apps/sidecar/src/workflow-substrate-factory/context-budget.ts @@ -1,40 +1,119 @@ -// Best-effort per-model context-window budget for the workbench -// director's compaction gate (CL-6204). +// Per-model context-window budget for the workbench director's +// compaction gate. // -// Reads the same `numCtx` override shape `@corbits/ollama-adapter` -// resolves against a live request (`OllamaAdapterConfig`'s -// `default`/`perModel` bag) directly off `InferenceSource.quirks`, so a -// deployment that already pins a per-model `num_ctx` for Ollama gets -// that same number as its compaction budget instead of a second, -// drifting constant. `quirks` is `unknown` at this boundary (an -// operator-authored, per-source passthrough bag), so every read here is -// defensive rather than a validating parse: an unrecognized shape (every -// non-Ollama source today) falls back to a conservative default sized -// for the smallest models this repo deploys against, rather than -// throwing or silently trusting a shape the adapter itself doesn't own -// at this call site. +// Window size is the resolved model's advertised native context, not a +// baked 8k-token (32k-char) cap: // -// Token counts are estimated from character counts (`CHARS_PER_TOKEN_ESTIMATE`) -// -- no tokenizer is available at this layer -- so the budget is -// deliberately conservative (see `COMPACTION_HEADROOM`), leaving room -// for the system prompt, tool definitions, and the model's own reply -// inside the same window. +// 1. `InferenceSource.quirks` -- the same `numCtx` override shape +// `@corbits/ollama-adapter` resolves (`OllamaAdapterConfig`'s +// `default`/`perModel` bag), plus a few other common token-window +// field names operators actually pin. +// 2. The pinned catalog's advertised native window for that model +// name (Ollama table entries and hosted-family prefixes). +// 3. A modern hosted-model fallback (128k tokens), used only when +// neither quirks nor the catalog name the model. Compact (see +// `COMPACTION_HEADROOM`) remains the safety net inside that +// window; `CONTEXT_OVERFLOW_MESSAGE` is reserved for the hard +// edge. +// +// `quirks` is `unknown` at this boundary (an operator-authored, +// per-source passthrough bag), so every read here is defensive rather +// than a validating parse. +// +// Token counts are estimated from character counts +// (`CHARS_PER_TOKEN_ESTIMATE`) -- no tokenizer is available at this +// layer -- so the budget is deliberately conservative (see +// `COMPACTION_HEADROOM`), leaving room for the system prompt, tool +// definitions, and the model's own reply inside the same window. -const DEFAULT_CONTEXT_BUDGET_TOKENS = 8_000; +const FALLBACK_CONTEXT_WINDOW_TOKENS = 128_000; const CHARS_PER_TOKEN_ESTIMATE = 4; const COMPACTION_HEADROOM = 0.6; -function readNumCtxFromBag(bag: Record): number | undefined { - const numCtx = bag.numCtx; - return typeof numCtx === "number" && numCtx > 0 ? numCtx : undefined; +// Exact names from `@corbits/inference-catalog`'s Ollama native-window +// table, then longest-prefix families for hosted catalog models. +// Longest prefix wins so `gpt-4.1` is not swallowed by `gpt-4`. +const ADVERTISED_CONTEXT_WINDOWS: readonly (readonly [string, number])[] = [ + ["qwen3.5:9b-mlx", 32_768], + ["qwen3.8:27b", 32_768], + ["llama3.1:8b", 131_072], + ["gpt-oss:20b", 131_072], + ["gpt-4-turbo", 128_000], + ["gpt-4.1", 1_047_576], + ["llama3.1", 131_072], + ["gpt-oss", 131_072], + ["gpt-4o", 128_000], + ["claude", 200_000], + ["gemini", 1_048_576], + ["gpt-5", 400_000], + ["gpt-4", 8_192], + ["grok", 256_000], + ["qwen3", 32_768], + ["llama", 131_072], + ["kimi", 128_000], + ["deepseek", 128_000], + ["minimax", 128_000], + ["glm", 128_000], + ["o1", 200_000], + ["o3", 200_000], + ["o4", 200_000], +]; + +function positiveTokenCount(value: unknown): number | undefined { + return typeof value === "number" && Number.isFinite(value) && value > 0 + ? value + : undefined; +} + +function readWindowFromBag(bag: Record): number | undefined { + return ( + positiveTokenCount(bag.numCtx) ?? + positiveTokenCount(bag.contextWindow) ?? + positiveTokenCount(bag.contextLength) ?? + positiveTokenCount(bag.maxContextTokens) + ); +} + +function catalogModelKey(model: string): string { + const trimmed = model.trim().toLowerCase(); + const slash = trimmed.lastIndexOf("/"); + return slash === -1 ? trimmed : trimmed.slice(slash + 1); +} + +/** + * Advertised native context window in tokens for a catalog model name. + * Exact Ollama table entries win; otherwise the longest matching hosted + * family prefix. `undefined` if this module has no published window for + * the name. + */ +export function advertisedContextWindowTokens( + model: string, +): number | undefined { + const key = catalogModelKey(model); + if (key.length === 0) { + return undefined; + } + let best: { prefixLength: number; tokens: number } | undefined; + for (const [prefix, tokens] of ADVERTISED_CONTEXT_WINDOWS) { + if (key === prefix) { + return tokens; + } + if ( + key.startsWith(prefix) && + (best === undefined || prefix.length > best.prefixLength) + ) { + best = { prefixLength: prefix.length, tokens }; + } + } + return best?.tokens; } /** - * Best-effort read of a per-model `numCtx` off an `InferenceSource.quirks` - * bag shaped like `@corbits/ollama-adapter`'s `OllamaAdapterConfig` - * (`{ default?: { numCtx? }, perModel?: { [model]: { numCtx? } } }`). - * Returns `undefined` for any other shape rather than throwing -- - * `quirks` is provider-specific and most sources carry none of this. + * Best-effort read of a per-model window hint off an + * `InferenceSource.quirks` bag. Understands `@corbits/ollama-adapter`'s + * `OllamaAdapterConfig` (`{ default?: { numCtx? }, perModel?: { [model]: + * { numCtx? } } }`) and a few other common token-window field names. + * Returns `undefined` for any unrecognized shape rather than throwing. */ export function readNumCtxHint( quirks: unknown, @@ -48,7 +127,7 @@ export function readNumCtxHint( if (typeof perModel === "object" && perModel !== null) { const entry = (perModel as Record)[model]; if (typeof entry === "object" && entry !== null) { - const fromPerModel = readNumCtxFromBag(entry as Record); + const fromPerModel = readWindowFromBag(entry as Record); if (fromPerModel !== undefined) { return fromPerModel; } @@ -56,23 +135,48 @@ export function readNumCtxHint( } const base = bag.default; if (typeof base === "object" && base !== null) { - return readNumCtxFromBag(base as Record); + const fromDefault = readWindowFromBag(base as Record); + if (fromDefault !== undefined) { + return fromDefault; + } } - return undefined; + return readWindowFromBag(bag); +} + +/** + * Resolved context window in tokens for one `InferenceSource`: quirks + * pin, then the advertised catalog window, then the hosted-model + * fallback. Compact headroom is applied by the char-budget helpers, + * not here. + */ +export function resolveContextWindowTokens( + quirks: unknown, + model: string, +): number { + return ( + readNumCtxHint(quirks, model) ?? + advertisedContextWindowTokens(model) ?? + FALLBACK_CONTEXT_WINDOW_TOKENS + ); +} + +function toChars(tokens: number, headroom: number): number { + return Math.floor(tokens * CHARS_PER_TOKEN_ESTIMATE * headroom); } /** * Resolve the compaction budget, in characters, for one `InferenceSource`. - * Applies `COMPACTION_HEADROOM` on top of the resolved (or default) - * `numCtx` so compaction fires with room left in the window, not at its - * hard edge. + * Applies `COMPACTION_HEADROOM` on top of the resolved window so + * compaction fires with room left, not at the hard edge. */ export function resolveContextBudgetChars( quirks: unknown, model: string, ): number { - const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS; - return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE * COMPACTION_HEADROOM); + return toChars( + resolveContextWindowTokens(quirks, model), + COMPACTION_HEADROOM, + ); } /** @@ -85,6 +189,5 @@ export function resolveHardContextLimitChars( quirks: unknown, model: string, ): number { - const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS; - return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE); + return toChars(resolveContextWindowTokens(quirks, model), 1); } diff --git a/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts b/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts index 80104b171..7349f29cd 100644 --- a/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts +++ b/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts @@ -248,6 +248,32 @@ test("context budget: a short conversation under budget is untouched (infers nor expect(typesOf(actions)).toEqual(["infer"]); }); +test("context budget: message.received over budget but under the hard limit compacts instead of inferring", async () => { + const director = createWorkbenchDirector( + "you are a test agent", + [], + {}, + { + budgetChars: 100, + hardLimitChars: 1_000_000, + compactorName: "summarize-budgeted-turns", + }, + ); + const overBudget = stateWithTurns([ + conversationTurn("a".repeat(200)), + conversationTurn("b".repeat(200)), + ]); + + const actions = await director.decide( + { type: "message.received", message: { id: "m1", content: "hi" } as never }, + overBudget, + caps, + ); + + expect(typesOf(actions)).toEqual(["checkpoint", "compact"]); + expect(replyOf(actions)).toBeUndefined(); +}); + test("context budget: tool-heavy history past the hard limit is caught even though every turn's text excerpt is short", async () => { // The reviewer's exact repro: 10 turns each carrying a 20,000-char // tool_result. Measured by placeholder length this was ~160 chars diff --git a/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts b/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts index e8362223c..fef82f875 100644 --- a/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts +++ b/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts @@ -1,5 +1,5 @@ // Workbench director: DefaultDirector plus an empty-turn retry and a -// context-budget gate (CL-6204). +// context-budget gate. // // `@intx/inference`'s DefaultDirector checkpoints and waits when // inference.done has no text and no tool calls. The human then sits in a @@ -15,27 +15,25 @@ // includes `infer` through, checks the turn history against the // model's real context window (`resolveContextBudgetChars` / // `resolveHardContextLimitChars` in `./context-budget`, sized from -// `InferenceSource.quirks`). +// `InferenceSource.quirks` and the advertised catalog window). // // - Over the hard limit (no headroom left at all): this is Ollama's // silent-truncation case made honest -- reply with the same // "exceeded the model's context limit" message hosted providers // produce via `inference.error`'s `context_overflow` category // (`@intx/inference`'s `default-director.js`), instead of sending a -// request that would truncate server-side with no error. -// - Over the (headroomed) budget but under the hard limit, on a -// non-final `tool.done` in a multi-call batch (DefaultDirector -// returns `[]` for every `tool.done` before the batch's last): -// fire `caps.compact` here instead of the no-op. This is the one -// point in the reactor's event flow where `compact` cannot stall a -// reply -- the batch's remaining `tool.done` events are already -// enqueued (`@intx/inference`'s `reactor.js` `executeTools` enqueues -// every result from one `Promise.all` before the reactor dequeues -// any of them) and will still drive the eventual re-infer. Firing -// compact on `message.received` or the batch's *last* `tool.done` -// would instead stall that message's reply waiting for a follow-up -// event the reactor never produces (`compactors.ts`'s header -// comment) -- this director never does that. +// request that would truncate server-side with no error. Compact +// cannot help here, so `CONTEXT_OVERFLOW_MESSAGE` is the only reply. +// - Over the (headroomed) budget but under the hard limit, on +// `message.received` or a non-final `tool.done` in a multi-call batch +// (DefaultDirector returns `[]` for every `tool.done` before the +// batch's last): fire `caps.compact` instead of inferring (or the +// no-op). Compacting on `message.received` keeps a later turn from +// inferring over a history already past budget. The batch's remaining +// `tool.done` events are already enqueued (`@intx/inference`'s +// `reactor.js` `executeTools` enqueues every result from one +// `Promise.all` before the reactor dequeues any of them) and will +// still drive the eventual re-infer. // - Otherwise: let the inner decision through unchanged. Compaction is // deferred to the next safe point rather than forced here. // @@ -218,13 +216,32 @@ export class WorkbenchDirector implements ReactorDirector { const list = Array.isArray(actions) ? actions : [actions]; const chars = estimateTurnsChars(state.turns); - if (list.some((action) => action.type === "infer")) { - if (chars > this.contextBudget.hardLimitChars) { + if (chars > this.contextBudget.hardLimitChars) { + if ( + list.some((action) => action.type === "infer") || + event.type === "message.received" + ) { return [ capabilities.checkpoint("context-overflow"), capabilities.reply(CONTEXT_OVERFLOW_MESSAGE), ]; } + } + + if ( + event.type === "message.received" && + chars > this.contextBudget.budgetChars + ) { + return [ + capabilities.checkpoint("context-budget-compact"), + capabilities.compact( + this.contextBudget.compactorName, + "context-budget", + ), + ]; + } + + if (list.some((action) => action.type === "infer")) { return undefined; } From 8d0687d734ba10643ee03f81651d8d60c8dcb6ea Mon Sep 17 00:00:00 2001 From: Sawyer Cutler Date: Fri, 28 Aug 2026 09:04:42 -0700 Subject: [PATCH 2/2] Infer on over-budget inbound instead of compacting only The reactor forbids compact+infer in one cycle and does not re-enter the director after compact. Replacing infer with compact on message.received left the user message unanswered until the next inbound event. Compact stays on the non-final tool.done path, which still re-infers. --- .../workbench-director.test.ts | 6 ++-- .../workbench-director.ts | 33 +++++++------------ 2 files changed, 14 insertions(+), 25 deletions(-) diff --git a/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts b/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts index 7349f29cd..c33d0ffaa 100644 --- a/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts +++ b/apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts @@ -248,7 +248,7 @@ test("context budget: a short conversation under budget is untouched (infers nor expect(typesOf(actions)).toEqual(["infer"]); }); -test("context budget: message.received over budget but under the hard limit compacts instead of inferring", async () => { +test("context budget: message.received over budget but under the hard limit still infers", async () => { const director = createWorkbenchDirector( "you are a test agent", [], @@ -270,8 +270,8 @@ test("context budget: message.received over budget but under the hard limit comp caps, ); - expect(typesOf(actions)).toEqual(["checkpoint", "compact"]); - expect(replyOf(actions)).toBeUndefined(); + expect(typesOf(actions)).toEqual(["infer"]); + expect(typesOf(actions)).not.toContain("compact"); }); test("context budget: tool-heavy history past the hard limit is caught even though every turn's text excerpt is short", async () => { diff --git a/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts b/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts index fef82f875..9bde5f143 100644 --- a/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts +++ b/apps/sidecar/src/workflow-substrate-factory/workbench-director.ts @@ -11,8 +11,7 @@ // We compose that path rather than reimplement it. // // Context budget (`contextBudget`, optional -- absent means unbudgeted, -// today's pre-CL-6204 behavior): before letting any inner decision that -// includes `infer` through, checks the turn history against the +// today's pre-CL-6204 behavior): checks the turn history against the // model's real context window (`resolveContextBudgetChars` / // `resolveHardContextLimitChars` in `./context-budget`, sized from // `InferenceSource.quirks` and the advertised catalog window). @@ -24,16 +23,19 @@ // (`@intx/inference`'s `default-director.js`), instead of sending a // request that would truncate server-side with no error. Compact // cannot help here, so `CONTEXT_OVERFLOW_MESSAGE` is the only reply. -// - Over the (headroomed) budget but under the hard limit, on -// `message.received` or a non-final `tool.done` in a multi-call batch -// (DefaultDirector returns `[]` for every `tool.done` before the -// batch's last): fire `caps.compact` instead of inferring (or the -// no-op). Compacting on `message.received` keeps a later turn from -// inferring over a history already past budget. The batch's remaining +// - Over the (headroomed) budget but under the hard limit, on a +// non-final `tool.done` in a multi-call batch (DefaultDirector +// returns `[]` for every `tool.done` before the batch's last): fire +// `caps.compact` instead of the no-op. The batch's remaining // `tool.done` events are already enqueued (`@intx/inference`'s // `reactor.js` `executeTools` enqueues every result from one // `Promise.all` before the reactor dequeues any of them) and will -// still drive the eventual re-infer. +// still drive the eventual re-infer. Do not compact as the sole +// terminal action on `message.received`: the reactor forbids +// compact+infer in one cycle and does not re-enter the director +// after compact, so compact-only would leave the inbound message +// unanswered. Over-budget inbound still infers; compaction waits +// for that next safe point. // - Otherwise: let the inner decision through unchanged. Compaction is // deferred to the next safe point rather than forced here. // @@ -228,19 +230,6 @@ export class WorkbenchDirector implements ReactorDirector { } } - if ( - event.type === "message.received" && - chars > this.contextBudget.budgetChars - ) { - return [ - capabilities.checkpoint("context-budget-compact"), - capabilities.compact( - this.contextBudget.compactorName, - "context-budget", - ), - ]; - } - if (list.some((action) => action.type === "infer")) { return undefined; }