Skip to content

Commit 08d1738

Browse files
Merge pull request #454 from corbitsdev/cl-7170-size-sidecar-context-budget-from-the-models-real-window
Size sidecar context budget from the model's real window
2 parents 583d892 + 8d0687d commit 08d1738

4 files changed

Lines changed: 238 additions & 61 deletions

File tree

‎apps/sidecar/src/workflow-substrate-factory/context-budget.test.ts‎

Lines changed: 47 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -1,12 +1,13 @@
1-
// Tests for the CL-6204 context-budget resolver: reads a per-model
2-
// `numCtx` hint off an `InferenceSource.quirks` bag shaped like
3-
// `@corbits/ollama-adapter`'s `OllamaAdapterConfig`, falling back to a
4-
// conservative default for any other shape.
1+
// Tests for the context-budget resolver: quirks `numCtx` pin, then the
2+
// advertised catalog window for the model name, never a baked 8k-token
3+
// (32k-char) cap for frontier models.
54
import { expect, test } from "bun:test";
65

76
import {
7+
advertisedContextWindowTokens,
88
readNumCtxHint,
99
resolveContextBudgetChars,
10+
resolveContextWindowTokens,
1011
resolveHardContextLimitChars,
1112
} from "./context-budget";
1213

@@ -27,6 +28,10 @@ test("readNumCtxHint: unrecognized or absent quirks resolve to undefined", () =>
2728
expect(readNumCtxHint("not-an-object", "claude")).toBeUndefined();
2829
});
2930

31+
test("readNumCtxHint: a top-level numCtx pin is visible without the Ollama bag shape", () => {
32+
expect(readNumCtxHint({ numCtx: 200_000 }, "claude-sonnet-5")).toBe(200_000);
33+
});
34+
3035
test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () => {
3136
const small = resolveContextBudgetChars(
3237
{ default: { numCtx: 32_000 } },
@@ -40,11 +45,48 @@ test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () =>
4045
expect(large).toBeGreaterThan(small);
4146
});
4247

43-
test("resolveContextBudgetChars: unknown model falls back to the conservative default", () => {
48+
test("resolveContextWindowTokens: quirks pin wins over the advertised catalog window", () => {
49+
expect(
50+
resolveContextWindowTokens(
51+
{ default: { numCtx: 8_192 } },
52+
"claude-sonnet-5",
53+
),
54+
).toBe(8_192);
55+
});
56+
57+
test("resolveContextWindowTokens: a frontier catalog model gets its advertised window, not a 32k-char cap", () => {
58+
const claudeTokens = resolveContextWindowTokens(undefined, "claude-sonnet-5");
59+
const claudeHard = resolveHardContextLimitChars(undefined, "claude-sonnet-5");
60+
const oldDefaultHardChars = 8_000 * 4;
61+
62+
expect(claudeTokens).toBe(200_000);
63+
expect(claudeHard).toBeGreaterThan(oldDefaultHardChars);
64+
expect(claudeHard).toBe(200_000 * 4);
65+
});
66+
67+
test("resolveContextWindowTokens: an Ollama catalog model gets its native window without quirks", () => {
68+
expect(resolveContextWindowTokens(undefined, "gpt-oss:20b")).toBe(131_072);
69+
expect(resolveContextWindowTokens(undefined, "qwen3.8:27b")).toBe(32_768);
70+
expect(
71+
resolveHardContextLimitChars(undefined, "gpt-oss:20b"),
72+
).toBeGreaterThan(resolveHardContextLimitChars(undefined, "qwen3.8:27b"));
73+
});
74+
75+
test("advertisedContextWindowTokens: a relay-prefixed name still matches the catalog model", () => {
76+
expect(advertisedContextWindowTokens("anthropic/claude-sonnet-5")).toBe(
77+
200_000,
78+
);
79+
expect(advertisedContextWindowTokens("openai/gpt-4.1")).toBe(1_047_576);
80+
});
81+
82+
test("resolveContextBudgetChars: unknown model falls back to a hosted-sized window, not 8k tokens", () => {
4483
const withoutQuirks = resolveContextBudgetChars(undefined, "unknown-model");
84+
const hard = resolveHardContextLimitChars(undefined, "unknown-model");
4585

4686
expect(withoutQuirks).toBeGreaterThan(0);
4787
expect(Number.isFinite(withoutQuirks)).toBe(true);
88+
expect(hard).toBeGreaterThan(8_000 * 4);
89+
expect(hard).toBe(128_000 * 4);
4890
});
4991

5092
test("resolveHardContextLimitChars: sits above the headroomed budget for the same source", () => {
Lines changed: 141 additions & 38 deletions
Original file line numberDiff line numberDiff line change
@@ -1,40 +1,119 @@
1-
// Best-effort per-model context-window budget for the workbench
2-
// director's compaction gate (CL-6204).
1+
// Per-model context-window budget for the workbench director's
2+
// compaction gate.
33
//
4-
// Reads the same `numCtx` override shape `@corbits/ollama-adapter`
5-
// resolves against a live request (`OllamaAdapterConfig`'s
6-
// `default`/`perModel` bag) directly off `InferenceSource.quirks`, so a
7-
// deployment that already pins a per-model `num_ctx` for Ollama gets
8-
// that same number as its compaction budget instead of a second,
9-
// drifting constant. `quirks` is `unknown` at this boundary (an
10-
// operator-authored, per-source passthrough bag), so every read here is
11-
// defensive rather than a validating parse: an unrecognized shape (every
12-
// non-Ollama source today) falls back to a conservative default sized
13-
// for the smallest models this repo deploys against, rather than
14-
// throwing or silently trusting a shape the adapter itself doesn't own
15-
// at this call site.
4+
// Window size is the resolved model's advertised native context, not a
5+
// baked 8k-token (32k-char) cap:
166
//
17-
// Token counts are estimated from character counts (`CHARS_PER_TOKEN_ESTIMATE`)
18-
// -- no tokenizer is available at this layer -- so the budget is
19-
// deliberately conservative (see `COMPACTION_HEADROOM`), leaving room
20-
// for the system prompt, tool definitions, and the model's own reply
21-
// inside the same window.
7+
// 1. `InferenceSource.quirks` -- the same `numCtx` override shape
8+
// `@corbits/ollama-adapter` resolves (`OllamaAdapterConfig`'s
9+
// `default`/`perModel` bag), plus a few other common token-window
10+
// field names operators actually pin.
11+
// 2. The pinned catalog's advertised native window for that model
12+
// name (Ollama table entries and hosted-family prefixes).
13+
// 3. A modern hosted-model fallback (128k tokens), used only when
14+
// neither quirks nor the catalog name the model. Compact (see
15+
// `COMPACTION_HEADROOM`) remains the safety net inside that
16+
// window; `CONTEXT_OVERFLOW_MESSAGE` is reserved for the hard
17+
// edge.
18+
//
19+
// `quirks` is `unknown` at this boundary (an operator-authored,
20+
// per-source passthrough bag), so every read here is defensive rather
21+
// than a validating parse.
22+
//
23+
// Token counts are estimated from character counts
24+
// (`CHARS_PER_TOKEN_ESTIMATE`) -- no tokenizer is available at this
25+
// layer -- so the budget is deliberately conservative (see
26+
// `COMPACTION_HEADROOM`), leaving room for the system prompt, tool
27+
// definitions, and the model's own reply inside the same window.
2228

23-
const DEFAULT_CONTEXT_BUDGET_TOKENS = 8_000;
29+
const FALLBACK_CONTEXT_WINDOW_TOKENS = 128_000;
2430
const CHARS_PER_TOKEN_ESTIMATE = 4;
2531
const COMPACTION_HEADROOM = 0.6;
2632

27-
function readNumCtxFromBag(bag: Record<string, unknown>): number | undefined {
28-
const numCtx = bag.numCtx;
29-
return typeof numCtx === "number" && numCtx > 0 ? numCtx : undefined;
33+
// Exact names from `@corbits/inference-catalog`'s Ollama native-window
34+
// table, then longest-prefix families for hosted catalog models.
35+
// Longest prefix wins so `gpt-4.1` is not swallowed by `gpt-4`.
36+
const ADVERTISED_CONTEXT_WINDOWS: readonly (readonly [string, number])[] = [
37+
["qwen3.5:9b-mlx", 32_768],
38+
["qwen3.8:27b", 32_768],
39+
["llama3.1:8b", 131_072],
40+
["gpt-oss:20b", 131_072],
41+
["gpt-4-turbo", 128_000],
42+
["gpt-4.1", 1_047_576],
43+
["llama3.1", 131_072],
44+
["gpt-oss", 131_072],
45+
["gpt-4o", 128_000],
46+
["claude", 200_000],
47+
["gemini", 1_048_576],
48+
["gpt-5", 400_000],
49+
["gpt-4", 8_192],
50+
["grok", 256_000],
51+
["qwen3", 32_768],
52+
["llama", 131_072],
53+
["kimi", 128_000],
54+
["deepseek", 128_000],
55+
["minimax", 128_000],
56+
["glm", 128_000],
57+
["o1", 200_000],
58+
["o3", 200_000],
59+
["o4", 200_000],
60+
];
61+
62+
function positiveTokenCount(value: unknown): number | undefined {
63+
return typeof value === "number" && Number.isFinite(value) && value > 0
64+
? value
65+
: undefined;
66+
}
67+
68+
function readWindowFromBag(bag: Record<string, unknown>): number | undefined {
69+
return (
70+
positiveTokenCount(bag.numCtx) ??
71+
positiveTokenCount(bag.contextWindow) ??
72+
positiveTokenCount(bag.contextLength) ??
73+
positiveTokenCount(bag.maxContextTokens)
74+
);
75+
}
76+
77+
function catalogModelKey(model: string): string {
78+
const trimmed = model.trim().toLowerCase();
79+
const slash = trimmed.lastIndexOf("/");
80+
return slash === -1 ? trimmed : trimmed.slice(slash + 1);
81+
}
82+
83+
/**
84+
* Advertised native context window in tokens for a catalog model name.
85+
* Exact Ollama table entries win; otherwise the longest matching hosted
86+
* family prefix. `undefined` if this module has no published window for
87+
* the name.
88+
*/
89+
export function advertisedContextWindowTokens(
90+
model: string,
91+
): number | undefined {
92+
const key = catalogModelKey(model);
93+
if (key.length === 0) {
94+
return undefined;
95+
}
96+
let best: { prefixLength: number; tokens: number } | undefined;
97+
for (const [prefix, tokens] of ADVERTISED_CONTEXT_WINDOWS) {
98+
if (key === prefix) {
99+
return tokens;
100+
}
101+
if (
102+
key.startsWith(prefix) &&
103+
(best === undefined || prefix.length > best.prefixLength)
104+
) {
105+
best = { prefixLength: prefix.length, tokens };
106+
}
107+
}
108+
return best?.tokens;
30109
}
31110

32111
/**
33-
* Best-effort read of a per-model `numCtx` off an `InferenceSource.quirks`
34-
* bag shaped like `@corbits/ollama-adapter`'s `OllamaAdapterConfig`
35-
* (`{ default?: { numCtx? }, perModel?: { [model]: { numCtx? } } }`).
36-
* Returns `undefined` for any other shape rather than throwing --
37-
* `quirks` is provider-specific and most sources carry none of this.
112+
* Best-effort read of a per-model window hint off an
113+
* `InferenceSource.quirks` bag. Understands `@corbits/ollama-adapter`'s
114+
* `OllamaAdapterConfig` (`{ default?: { numCtx? }, perModel?: { [model]:
115+
* { numCtx? } } }`) and a few other common token-window field names.
116+
* Returns `undefined` for any unrecognized shape rather than throwing.
38117
*/
39118
export function readNumCtxHint(
40119
quirks: unknown,
@@ -48,31 +127,56 @@ export function readNumCtxHint(
48127
if (typeof perModel === "object" && perModel !== null) {
49128
const entry = (perModel as Record<string, unknown>)[model];
50129
if (typeof entry === "object" && entry !== null) {
51-
const fromPerModel = readNumCtxFromBag(entry as Record<string, unknown>);
130+
const fromPerModel = readWindowFromBag(entry as Record<string, unknown>);
52131
if (fromPerModel !== undefined) {
53132
return fromPerModel;
54133
}
55134
}
56135
}
57136
const base = bag.default;
58137
if (typeof base === "object" && base !== null) {
59-
return readNumCtxFromBag(base as Record<string, unknown>);
138+
const fromDefault = readWindowFromBag(base as Record<string, unknown>);
139+
if (fromDefault !== undefined) {
140+
return fromDefault;
141+
}
60142
}
61-
return undefined;
143+
return readWindowFromBag(bag);
144+
}
145+
146+
/**
147+
* Resolved context window in tokens for one `InferenceSource`: quirks
148+
* pin, then the advertised catalog window, then the hosted-model
149+
* fallback. Compact headroom is applied by the char-budget helpers,
150+
* not here.
151+
*/
152+
export function resolveContextWindowTokens(
153+
quirks: unknown,
154+
model: string,
155+
): number {
156+
return (
157+
readNumCtxHint(quirks, model) ??
158+
advertisedContextWindowTokens(model) ??
159+
FALLBACK_CONTEXT_WINDOW_TOKENS
160+
);
161+
}
162+
163+
function toChars(tokens: number, headroom: number): number {
164+
return Math.floor(tokens * CHARS_PER_TOKEN_ESTIMATE * headroom);
62165
}
63166

64167
/**
65168
* Resolve the compaction budget, in characters, for one `InferenceSource`.
66-
* Applies `COMPACTION_HEADROOM` on top of the resolved (or default)
67-
* `numCtx` so compaction fires with room left in the window, not at its
68-
* hard edge.
169+
* Applies `COMPACTION_HEADROOM` on top of the resolved window so
170+
* compaction fires with room left, not at the hard edge.
69171
*/
70172
export function resolveContextBudgetChars(
71173
quirks: unknown,
72174
model: string,
73175
): number {
74-
const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS;
75-
return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE * COMPACTION_HEADROOM);
176+
return toChars(
177+
resolveContextWindowTokens(quirks, model),
178+
COMPACTION_HEADROOM,
179+
);
76180
}
77181

78182
/**
@@ -85,6 +189,5 @@ export function resolveHardContextLimitChars(
85189
quirks: unknown,
86190
model: string,
87191
): number {
88-
const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS;
89-
return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE);
192+
return toChars(resolveContextWindowTokens(quirks, model), 1);
90193
}

‎apps/sidecar/src/workflow-substrate-factory/workbench-director.test.ts‎

Lines changed: 26 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -248,6 +248,32 @@ test("context budget: a short conversation under budget is untouched (infers nor
248248
expect(typesOf(actions)).toEqual(["infer"]);
249249
});
250250

251+
test("context budget: message.received over budget but under the hard limit still infers", async () => {
252+
const director = createWorkbenchDirector(
253+
"you are a test agent",
254+
[],
255+
{},
256+
{
257+
budgetChars: 100,
258+
hardLimitChars: 1_000_000,
259+
compactorName: "summarize-budgeted-turns",
260+
},
261+
);
262+
const overBudget = stateWithTurns([
263+
conversationTurn("a".repeat(200)),
264+
conversationTurn("b".repeat(200)),
265+
]);
266+
267+
const actions = await director.decide(
268+
{ type: "message.received", message: { id: "m1", content: "hi" } as never },
269+
overBudget,
270+
caps,
271+
);
272+
273+
expect(typesOf(actions)).toEqual(["infer"]);
274+
expect(typesOf(actions)).not.toContain("compact");
275+
});
276+
251277
test("context budget: tool-heavy history past the hard limit is caught even though every turn's text excerpt is short", async () => {
252278
// The reviewer's exact repro: 10 turns each carrying a 20,000-char
253279
// tool_result. Measured by placeholder length this was ~160 chars

0 commit comments

Comments
 (0)