1- // Best-effort per- model context-window budget for the workbench
2- // director's compaction gate (CL-6204) .
1+ // Per- model context-window budget for the workbench director's
2+ // compaction gate.
33//
4- // Reads the same `numCtx` override shape `@corbits/ollama-adapter`
5- // resolves against a live request (`OllamaAdapterConfig`'s
6- // `default`/`perModel` bag) directly off `InferenceSource.quirks`, so a
7- // deployment that already pins a per-model `num_ctx` for Ollama gets
8- // that same number as its compaction budget instead of a second,
9- // drifting constant. `quirks` is `unknown` at this boundary (an
10- // operator-authored, per-source passthrough bag), so every read here is
11- // defensive rather than a validating parse: an unrecognized shape (every
12- // non-Ollama source today) falls back to a conservative default sized
13- // for the smallest models this repo deploys against, rather than
14- // throwing or silently trusting a shape the adapter itself doesn't own
15- // at this call site.
4+ // Window size is the resolved model's advertised native context, not a
5+ // baked 8k-token (32k-char) cap:
166//
17- // Token counts are estimated from character counts (`CHARS_PER_TOKEN_ESTIMATE`)
18- // -- no tokenizer is available at this layer -- so the budget is
19- // deliberately conservative (see `COMPACTION_HEADROOM`), leaving room
20- // for the system prompt, tool definitions, and the model's own reply
21- // inside the same window.
7+ // 1. `InferenceSource.quirks` -- the same `numCtx` override shape
8+ // `@corbits/ollama-adapter` resolves (`OllamaAdapterConfig`'s
9+ // `default`/`perModel` bag), plus a few other common token-window
10+ // field names operators actually pin.
11+ // 2. The pinned catalog's advertised native window for that model
12+ // name (Ollama table entries and hosted-family prefixes).
13+ // 3. A modern hosted-model fallback (128k tokens), used only when
14+ // neither quirks nor the catalog name the model. Compact (see
15+ // `COMPACTION_HEADROOM`) remains the safety net inside that
16+ // window; `CONTEXT_OVERFLOW_MESSAGE` is reserved for the hard
17+ // edge.
18+ //
19+ // `quirks` is `unknown` at this boundary (an operator-authored,
20+ // per-source passthrough bag), so every read here is defensive rather
21+ // than a validating parse.
22+ //
23+ // Token counts are estimated from character counts
24+ // (`CHARS_PER_TOKEN_ESTIMATE`) -- no tokenizer is available at this
25+ // layer -- so the budget is deliberately conservative (see
26+ // `COMPACTION_HEADROOM`), leaving room for the system prompt, tool
27+ // definitions, and the model's own reply inside the same window.
2228
23- const DEFAULT_CONTEXT_BUDGET_TOKENS = 8_000 ;
29+ const FALLBACK_CONTEXT_WINDOW_TOKENS = 128_000 ;
2430const CHARS_PER_TOKEN_ESTIMATE = 4 ;
2531const COMPACTION_HEADROOM = 0.6 ;
2632
27- function readNumCtxFromBag ( bag : Record < string , unknown > ) : number | undefined {
28- const numCtx = bag . numCtx ;
29- return typeof numCtx === "number" && numCtx > 0 ? numCtx : undefined ;
33+ // Exact names from `@corbits/inference-catalog`'s Ollama native-window
34+ // table, then longest-prefix families for hosted catalog models.
35+ // Longest prefix wins so `gpt-4.1` is not swallowed by `gpt-4`.
36+ const ADVERTISED_CONTEXT_WINDOWS : readonly ( readonly [ string , number ] ) [ ] = [
37+ [ "qwen3.5:9b-mlx" , 32_768 ] ,
38+ [ "qwen3.8:27b" , 32_768 ] ,
39+ [ "llama3.1:8b" , 131_072 ] ,
40+ [ "gpt-oss:20b" , 131_072 ] ,
41+ [ "gpt-4-turbo" , 128_000 ] ,
42+ [ "gpt-4.1" , 1_047_576 ] ,
43+ [ "llama3.1" , 131_072 ] ,
44+ [ "gpt-oss" , 131_072 ] ,
45+ [ "gpt-4o" , 128_000 ] ,
46+ [ "claude" , 200_000 ] ,
47+ [ "gemini" , 1_048_576 ] ,
48+ [ "gpt-5" , 400_000 ] ,
49+ [ "gpt-4" , 8_192 ] ,
50+ [ "grok" , 256_000 ] ,
51+ [ "qwen3" , 32_768 ] ,
52+ [ "llama" , 131_072 ] ,
53+ [ "kimi" , 128_000 ] ,
54+ [ "deepseek" , 128_000 ] ,
55+ [ "minimax" , 128_000 ] ,
56+ [ "glm" , 128_000 ] ,
57+ [ "o1" , 200_000 ] ,
58+ [ "o3" , 200_000 ] ,
59+ [ "o4" , 200_000 ] ,
60+ ] ;
61+
62+ function positiveTokenCount ( value : unknown ) : number | undefined {
63+ return typeof value === "number" && Number . isFinite ( value ) && value > 0
64+ ? value
65+ : undefined ;
66+ }
67+
68+ function readWindowFromBag ( bag : Record < string , unknown > ) : number | undefined {
69+ return (
70+ positiveTokenCount ( bag . numCtx ) ??
71+ positiveTokenCount ( bag . contextWindow ) ??
72+ positiveTokenCount ( bag . contextLength ) ??
73+ positiveTokenCount ( bag . maxContextTokens )
74+ ) ;
75+ }
76+
77+ function catalogModelKey ( model : string ) : string {
78+ const trimmed = model . trim ( ) . toLowerCase ( ) ;
79+ const slash = trimmed . lastIndexOf ( "/" ) ;
80+ return slash === - 1 ? trimmed : trimmed . slice ( slash + 1 ) ;
81+ }
82+
83+ /**
84+ * Advertised native context window in tokens for a catalog model name.
85+ * Exact Ollama table entries win; otherwise the longest matching hosted
86+ * family prefix. `undefined` if this module has no published window for
87+ * the name.
88+ */
89+ export function advertisedContextWindowTokens (
90+ model : string ,
91+ ) : number | undefined {
92+ const key = catalogModelKey ( model ) ;
93+ if ( key . length === 0 ) {
94+ return undefined ;
95+ }
96+ let best : { prefixLength : number ; tokens : number } | undefined ;
97+ for ( const [ prefix , tokens ] of ADVERTISED_CONTEXT_WINDOWS ) {
98+ if ( key === prefix ) {
99+ return tokens ;
100+ }
101+ if (
102+ key . startsWith ( prefix ) &&
103+ ( best === undefined || prefix . length > best . prefixLength )
104+ ) {
105+ best = { prefixLength : prefix . length , tokens } ;
106+ }
107+ }
108+ return best ?. tokens ;
30109}
31110
32111/**
33- * Best-effort read of a per-model `numCtx` off an `InferenceSource.quirks`
34- * bag shaped like `@corbits/ollama-adapter`'s `OllamaAdapterConfig`
35- * (`{ default?: { numCtx? }, perModel?: { [model]: { numCtx? } } }`).
36- * Returns `undefined` for any other shape rather than throwing --
37- * `quirks` is provider-specific and most sources carry none of this .
112+ * Best-effort read of a per-model window hint off an
113+ * `InferenceSource.quirks` bag. Understands `@corbits/ollama-adapter`'s
114+ * `OllamaAdapterConfig` (`{ default?: { numCtx? }, perModel?: { [model]:
115+ * { numCtx? } } }`) and a few other common token-window field names.
116+ * Returns `undefined` for any unrecognized shape rather than throwing .
38117 */
39118export function readNumCtxHint (
40119 quirks : unknown ,
@@ -48,31 +127,56 @@ export function readNumCtxHint(
48127 if ( typeof perModel === "object" && perModel !== null ) {
49128 const entry = ( perModel as Record < string , unknown > ) [ model ] ;
50129 if ( typeof entry === "object" && entry !== null ) {
51- const fromPerModel = readNumCtxFromBag ( entry as Record < string , unknown > ) ;
130+ const fromPerModel = readWindowFromBag ( entry as Record < string , unknown > ) ;
52131 if ( fromPerModel !== undefined ) {
53132 return fromPerModel ;
54133 }
55134 }
56135 }
57136 const base = bag . default ;
58137 if ( typeof base === "object" && base !== null ) {
59- return readNumCtxFromBag ( base as Record < string , unknown > ) ;
138+ const fromDefault = readWindowFromBag ( base as Record < string , unknown > ) ;
139+ if ( fromDefault !== undefined ) {
140+ return fromDefault ;
141+ }
60142 }
61- return undefined ;
143+ return readWindowFromBag ( bag ) ;
144+ }
145+
146+ /**
147+ * Resolved context window in tokens for one `InferenceSource`: quirks
148+ * pin, then the advertised catalog window, then the hosted-model
149+ * fallback. Compact headroom is applied by the char-budget helpers,
150+ * not here.
151+ */
152+ export function resolveContextWindowTokens (
153+ quirks : unknown ,
154+ model : string ,
155+ ) : number {
156+ return (
157+ readNumCtxHint ( quirks , model ) ??
158+ advertisedContextWindowTokens ( model ) ??
159+ FALLBACK_CONTEXT_WINDOW_TOKENS
160+ ) ;
161+ }
162+
163+ function toChars ( tokens : number , headroom : number ) : number {
164+ return Math . floor ( tokens * CHARS_PER_TOKEN_ESTIMATE * headroom ) ;
62165}
63166
64167/**
65168 * Resolve the compaction budget, in characters, for one `InferenceSource`.
66- * Applies `COMPACTION_HEADROOM` on top of the resolved (or default)
67- * `numCtx` so compaction fires with room left in the window, not at its
68- * hard edge.
169+ * Applies `COMPACTION_HEADROOM` on top of the resolved window so
170+ * compaction fires with room left, not at the hard edge.
69171 */
70172export function resolveContextBudgetChars (
71173 quirks : unknown ,
72174 model : string ,
73175) : number {
74- const numCtx = readNumCtxHint ( quirks , model ) ?? DEFAULT_CONTEXT_BUDGET_TOKENS ;
75- return Math . floor ( numCtx * CHARS_PER_TOKEN_ESTIMATE * COMPACTION_HEADROOM ) ;
176+ return toChars (
177+ resolveContextWindowTokens ( quirks , model ) ,
178+ COMPACTION_HEADROOM ,
179+ ) ;
76180}
77181
78182/**
@@ -85,6 +189,5 @@ export function resolveHardContextLimitChars(
85189 quirks : unknown ,
86190 model : string ,
87191) : number {
88- const numCtx = readNumCtxHint ( quirks , model ) ?? DEFAULT_CONTEXT_BUDGET_TOKENS ;
89- return Math . floor ( numCtx * CHARS_PER_TOKEN_ESTIMATE ) ;
192+ return toChars ( resolveContextWindowTokens ( quirks , model ) , 1 ) ;
90193}
0 commit comments