Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -1,12 +1,13 @@
// Tests for the CL-6204 context-budget resolver: reads a per-model
// `numCtx` hint off an `InferenceSource.quirks` bag shaped like
// `@corbits/ollama-adapter`'s `OllamaAdapterConfig`, falling back to a
// conservative default for any other shape.
// Tests for the context-budget resolver: quirks `numCtx` pin, then the
// advertised catalog window for the model name, never a baked 8k-token
// (32k-char) cap for frontier models.
import { expect, test } from "bun:test";

import {
advertisedContextWindowTokens,
readNumCtxHint,
resolveContextBudgetChars,
resolveContextWindowTokens,
resolveHardContextLimitChars,
} from "./context-budget";

Expand All @@ -27,6 +28,10 @@ test("readNumCtxHint: unrecognized or absent quirks resolve to undefined", () =>
expect(readNumCtxHint("not-an-object", "claude")).toBeUndefined();
});

test("readNumCtxHint: a top-level numCtx pin is visible without the Ollama bag shape", () => {
expect(readNumCtxHint({ numCtx: 200_000 }, "claude-sonnet-5")).toBe(200_000);
});

test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () => {
const small = resolveContextBudgetChars(
{ default: { numCtx: 32_000 } },
Expand All @@ -40,11 +45,48 @@ test("resolveContextBudgetChars: a bigger numCtx yields a bigger budget", () =>
expect(large).toBeGreaterThan(small);
});

test("resolveContextBudgetChars: unknown model falls back to the conservative default", () => {
test("resolveContextWindowTokens: quirks pin wins over the advertised catalog window", () => {
expect(
resolveContextWindowTokens(
{ default: { numCtx: 8_192 } },
"claude-sonnet-5",
),
).toBe(8_192);
});

test("resolveContextWindowTokens: a frontier catalog model gets its advertised window, not a 32k-char cap", () => {
const claudeTokens = resolveContextWindowTokens(undefined, "claude-sonnet-5");
const claudeHard = resolveHardContextLimitChars(undefined, "claude-sonnet-5");
const oldDefaultHardChars = 8_000 * 4;

expect(claudeTokens).toBe(200_000);
expect(claudeHard).toBeGreaterThan(oldDefaultHardChars);
expect(claudeHard).toBe(200_000 * 4);
});

test("resolveContextWindowTokens: an Ollama catalog model gets its native window without quirks", () => {
expect(resolveContextWindowTokens(undefined, "gpt-oss:20b")).toBe(131_072);
expect(resolveContextWindowTokens(undefined, "qwen3.8:27b")).toBe(32_768);
expect(
resolveHardContextLimitChars(undefined, "gpt-oss:20b"),
).toBeGreaterThan(resolveHardContextLimitChars(undefined, "qwen3.8:27b"));
});

test("advertisedContextWindowTokens: a relay-prefixed name still matches the catalog model", () => {
expect(advertisedContextWindowTokens("anthropic/claude-sonnet-5")).toBe(
200_000,
);
expect(advertisedContextWindowTokens("openai/gpt-4.1")).toBe(1_047_576);
});

test("resolveContextBudgetChars: unknown model falls back to a hosted-sized window, not 8k tokens", () => {
const withoutQuirks = resolveContextBudgetChars(undefined, "unknown-model");
const hard = resolveHardContextLimitChars(undefined, "unknown-model");

expect(withoutQuirks).toBeGreaterThan(0);
expect(Number.isFinite(withoutQuirks)).toBe(true);
expect(hard).toBeGreaterThan(8_000 * 4);
expect(hard).toBe(128_000 * 4);
});

test("resolveHardContextLimitChars: sits above the headroomed budget for the same source", () => {
Expand Down
179 changes: 141 additions & 38 deletions apps/sidecar/src/workflow-substrate-factory/context-budget.ts
Original file line number Diff line number Diff line change
@@ -1,40 +1,119 @@
// Best-effort per-model context-window budget for the workbench
// director's compaction gate (CL-6204).
// Per-model context-window budget for the workbench director's
// compaction gate.
//
// Reads the same `numCtx` override shape `@corbits/ollama-adapter`
// resolves against a live request (`OllamaAdapterConfig`'s
// `default`/`perModel` bag) directly off `InferenceSource.quirks`, so a
// deployment that already pins a per-model `num_ctx` for Ollama gets
// that same number as its compaction budget instead of a second,
// drifting constant. `quirks` is `unknown` at this boundary (an
// operator-authored, per-source passthrough bag), so every read here is
// defensive rather than a validating parse: an unrecognized shape (every
// non-Ollama source today) falls back to a conservative default sized
// for the smallest models this repo deploys against, rather than
// throwing or silently trusting a shape the adapter itself doesn't own
// at this call site.
// Window size is the resolved model's advertised native context, not a
// baked 8k-token (32k-char) cap:
//
// Token counts are estimated from character counts (`CHARS_PER_TOKEN_ESTIMATE`)
// -- no tokenizer is available at this layer -- so the budget is
// deliberately conservative (see `COMPACTION_HEADROOM`), leaving room
// for the system prompt, tool definitions, and the model's own reply
// inside the same window.
// 1. `InferenceSource.quirks` -- the same `numCtx` override shape
// `@corbits/ollama-adapter` resolves (`OllamaAdapterConfig`'s
// `default`/`perModel` bag), plus a few other common token-window
// field names operators actually pin.
// 2. The pinned catalog's advertised native window for that model
// name (Ollama table entries and hosted-family prefixes).
// 3. A modern hosted-model fallback (128k tokens), used only when
// neither quirks nor the catalog name the model. Compact (see
// `COMPACTION_HEADROOM`) remains the safety net inside that
// window; `CONTEXT_OVERFLOW_MESSAGE` is reserved for the hard
// edge.
//
// `quirks` is `unknown` at this boundary (an operator-authored,
// per-source passthrough bag), so every read here is defensive rather
// than a validating parse.
//
// Token counts are estimated from character counts
// (`CHARS_PER_TOKEN_ESTIMATE`) -- no tokenizer is available at this
// layer -- so the budget is deliberately conservative (see
// `COMPACTION_HEADROOM`), leaving room for the system prompt, tool
// definitions, and the model's own reply inside the same window.

const DEFAULT_CONTEXT_BUDGET_TOKENS = 8_000;
const FALLBACK_CONTEXT_WINDOW_TOKENS = 128_000;
const CHARS_PER_TOKEN_ESTIMATE = 4;
const COMPACTION_HEADROOM = 0.6;

function readNumCtxFromBag(bag: Record<string, unknown>): number | undefined {
const numCtx = bag.numCtx;
return typeof numCtx === "number" && numCtx > 0 ? numCtx : undefined;
// Exact names from `@corbits/inference-catalog`'s Ollama native-window
// table, then longest-prefix families for hosted catalog models.
// Longest prefix wins so `gpt-4.1` is not swallowed by `gpt-4`.
const ADVERTISED_CONTEXT_WINDOWS: readonly (readonly [string, number])[] = [
["qwen3.5:9b-mlx", 32_768],
["qwen3.8:27b", 32_768],
["llama3.1:8b", 131_072],
["gpt-oss:20b", 131_072],
["gpt-4-turbo", 128_000],
["gpt-4.1", 1_047_576],
["llama3.1", 131_072],
["gpt-oss", 131_072],
["gpt-4o", 128_000],
["claude", 200_000],
["gemini", 1_048_576],
["gpt-5", 400_000],
["gpt-4", 8_192],
["grok", 256_000],
["qwen3", 32_768],
["llama", 131_072],
["kimi", 128_000],
["deepseek", 128_000],
["minimax", 128_000],
["glm", 128_000],
["o1", 200_000],
["o3", 200_000],
["o4", 200_000],
];

function positiveTokenCount(value: unknown): number | undefined {
return typeof value === "number" && Number.isFinite(value) && value > 0
? value
: undefined;
}

function readWindowFromBag(bag: Record<string, unknown>): number | undefined {
return (
positiveTokenCount(bag.numCtx) ??
positiveTokenCount(bag.contextWindow) ??
positiveTokenCount(bag.contextLength) ??
positiveTokenCount(bag.maxContextTokens)
);
}

function catalogModelKey(model: string): string {
const trimmed = model.trim().toLowerCase();
const slash = trimmed.lastIndexOf("/");
return slash === -1 ? trimmed : trimmed.slice(slash + 1);
}

/**
* Advertised native context window in tokens for a catalog model name.
* Exact Ollama table entries win; otherwise the longest matching hosted
* family prefix. `undefined` if this module has no published window for
* the name.
*/
export function advertisedContextWindowTokens(
model: string,
): number | undefined {
const key = catalogModelKey(model);
if (key.length === 0) {
return undefined;
}
let best: { prefixLength: number; tokens: number } | undefined;
for (const [prefix, tokens] of ADVERTISED_CONTEXT_WINDOWS) {
if (key === prefix) {
return tokens;
}
if (
key.startsWith(prefix) &&
(best === undefined || prefix.length > best.prefixLength)
) {
best = { prefixLength: prefix.length, tokens };
}
}
return best?.tokens;
}

/**
* Best-effort read of a per-model `numCtx` off an `InferenceSource.quirks`
* bag shaped like `@corbits/ollama-adapter`'s `OllamaAdapterConfig`
* (`{ default?: { numCtx? }, perModel?: { [model]: { numCtx? } } }`).
* Returns `undefined` for any other shape rather than throwing --
* `quirks` is provider-specific and most sources carry none of this.
* Best-effort read of a per-model window hint off an
* `InferenceSource.quirks` bag. Understands `@corbits/ollama-adapter`'s
* `OllamaAdapterConfig` (`{ default?: { numCtx? }, perModel?: { [model]:
* { numCtx? } } }`) and a few other common token-window field names.
* Returns `undefined` for any unrecognized shape rather than throwing.
*/
export function readNumCtxHint(
quirks: unknown,
Expand All @@ -48,31 +127,56 @@ export function readNumCtxHint(
if (typeof perModel === "object" && perModel !== null) {
const entry = (perModel as Record<string, unknown>)[model];
if (typeof entry === "object" && entry !== null) {
const fromPerModel = readNumCtxFromBag(entry as Record<string, unknown>);
const fromPerModel = readWindowFromBag(entry as Record<string, unknown>);
if (fromPerModel !== undefined) {
return fromPerModel;
}
}
}
const base = bag.default;
if (typeof base === "object" && base !== null) {
return readNumCtxFromBag(base as Record<string, unknown>);
const fromDefault = readWindowFromBag(base as Record<string, unknown>);
if (fromDefault !== undefined) {
return fromDefault;
}
}
return undefined;
return readWindowFromBag(bag);
}

/**
* Resolved context window in tokens for one `InferenceSource`: quirks
* pin, then the advertised catalog window, then the hosted-model
* fallback. Compact headroom is applied by the char-budget helpers,
* not here.
*/
export function resolveContextWindowTokens(
quirks: unknown,
model: string,
): number {
return (
readNumCtxHint(quirks, model) ??
advertisedContextWindowTokens(model) ??
FALLBACK_CONTEXT_WINDOW_TOKENS
);
}

function toChars(tokens: number, headroom: number): number {
return Math.floor(tokens * CHARS_PER_TOKEN_ESTIMATE * headroom);
}

/**
* Resolve the compaction budget, in characters, for one `InferenceSource`.
* Applies `COMPACTION_HEADROOM` on top of the resolved (or default)
* `numCtx` so compaction fires with room left in the window, not at its
* hard edge.
* Applies `COMPACTION_HEADROOM` on top of the resolved window so
* compaction fires with room left, not at the hard edge.
*/
export function resolveContextBudgetChars(
quirks: unknown,
model: string,
): number {
const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS;
return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE * COMPACTION_HEADROOM);
return toChars(
resolveContextWindowTokens(quirks, model),
COMPACTION_HEADROOM,
);
}

/**
Expand All @@ -85,6 +189,5 @@ export function resolveHardContextLimitChars(
quirks: unknown,
model: string,
): number {
const numCtx = readNumCtxHint(quirks, model) ?? DEFAULT_CONTEXT_BUDGET_TOKENS;
return Math.floor(numCtx * CHARS_PER_TOKEN_ESTIMATE);
return toChars(resolveContextWindowTokens(quirks, model), 1);
}
Original file line number Diff line number Diff line change
Expand Up @@ -248,6 +248,32 @@ test("context budget: a short conversation under budget is untouched (infers nor
expect(typesOf(actions)).toEqual(["infer"]);
});

test("context budget: message.received over budget but under the hard limit still infers", async () => {
const director = createWorkbenchDirector(
"you are a test agent",
[],
{},
{
budgetChars: 100,
hardLimitChars: 1_000_000,
compactorName: "summarize-budgeted-turns",
},
);
const overBudget = stateWithTurns([
conversationTurn("a".repeat(200)),
conversationTurn("b".repeat(200)),
]);

const actions = await director.decide(
{ type: "message.received", message: { id: "m1", content: "hi" } as never },
overBudget,
caps,
);

expect(typesOf(actions)).toEqual(["infer"]);
expect(typesOf(actions)).not.toContain("compact");
});

test("context budget: tool-heavy history past the hard limit is caught even though every turn's text excerpt is short", async () => {
// The reviewer's exact repro: 10 turns each carrying a 20,000-char
// tool_result. Measured by placeholder length this was ~160 chars
Expand Down
Loading
Loading