From f63431936322b4217fd4483af791a42610718403 Mon Sep 17 00:00:00 2001 From: aurorax-neo <15047150695@163.com> Date: Fri, 25 Sep 2026 13:45:48 +0800 Subject: [PATCH 1/5] fix model catalog matching for routed ids --- .../electron/main/models-dev-catalog.ts | 181 +++++-------- apps/desktop/electron/main/plugin-mcp.ts | 30 ++- .../src/features/chat/composer/model.ts | 18 +- apps/desktop/src/lib/composer-models.ts | 4 +- .../composer-model-thinking-menu.test.mjs | 2 +- apps/desktop/test/composer-models.test.mjs | 24 +- apps/desktop/test/models-dev-catalog.test.mjs | 253 +++++++++++------- apps/desktop/test/plugin-mcp.test.mjs | 21 +- packages/shared/src/thinking-levels.test.ts | 6 +- packages/shared/src/thinking-levels.ts | 11 +- packages/shared/src/types/models.ts | 53 ++-- 11 files changed, 318 insertions(+), 285 deletions(-) diff --git a/apps/desktop/electron/main/models-dev-catalog.ts b/apps/desktop/electron/main/models-dev-catalog.ts index 972db9c104..8788b9fe92 100644 --- a/apps/desktop/electron/main/models-dev-catalog.ts +++ b/apps/desktop/electron/main/models-dev-catalog.ts @@ -899,56 +899,13 @@ class ModelsDevLookupIndex { } } -/** Index only aliases the matcher can accept: exact IDs, the region-stripped - * form, full slash-path leaves, known-vendor dash/dot prefixes, the supported - * thinking/agent/latest variants, and published release stamps. The matcher - * remains the final authority (including known-vendor conflicts), but the index - * must register every alias the matcher accepts: a key it never indexed is a - * lookup that can never reach the record. */ +/** Generate the exact lowercase last `/`-segment used by the catalog matcher. */ function candidateKeys(modelId: string): string[] { const normalized = normalizedModelId(modelId); if (!normalized) return []; - const keys = new Set(); - const at = normalized.indexOf("@"); - const base = at > 0 ? normalized.slice(0, at) : normalized; - - const add = (value: string) => { - if (!value) return; - keys.add(value); - const slash = value.lastIndexOf("/"); - if (slash >= 0 && slash < value.length - 1) keys.add(value.slice(slash + 1)); - for (const separator of ["-", "."] as const) { - for (const vendor of MODEL_VENDOR_PREFIXES) { - const head = `${vendor}${separator}`; - if (value.startsWith(head) && value.length > head.length) { - keys.add(value.slice(head.length)); - } - } - } - }; - - /* - Suffixes are stripped after the region, just as `catalogModelIdsMatch` - does, and either suffix may be the outer one: `foo-0731-thinking` and - `foo-thinking-0731` both reach `foo`. Aliases are added from both sides so a - suffixed catalog id and a suffixed request each find the other. - */ - const variants = new Set(); - for (const value of [normalized, base]) { - const withoutVariant = stripVariantSuffix(value); - const withoutRelease = stripReleaseSuffix(value); - for (const variant of [ - value, - withoutVariant, - withoutRelease, - stripReleaseSuffix(withoutVariant), - stripVariantSuffix(withoutRelease), - ]) { - variants.add(variant); - } - } - for (const value of variants) add(value); - return [...keys]; + const slash = normalized.lastIndexOf("/"); + const leaf = slash >= 0 ? normalized.slice(slash + 1) : normalized; + return leaf ? [leaf] : []; } function registrationKeys(modelId: string): string[] { @@ -961,7 +918,54 @@ function lookupCandidateKeys(requested: string): string[] { const EMPTY_CANDIDATES: readonly IndexedModel[] = []; +type OfficialProviderFamily = "anthropic" | "openai" | "google" | "xai"; + +const OFFICIAL_PROVIDER_FAMILIES: Readonly> = { + anthropic: "anthropic", + openai: "openai", + google: "google", + "google-ai-studio": "google", + "google-vertex": "google", + xai: "xai", + "x-ai": "xai", +}; + +function explicitModelSourceFamily(modelId: string): OfficialProviderFamily | undefined { + const firstSegment = normalizedModelId(modelId).split("/", 1)[0]; + return OFFICIAL_PROVIDER_FAMILIES[firstSegment]; +} + +function isOfficialSourceProvider(entry: IndexedModel): boolean { + const providerFamily = OFFICIAL_PROVIDER_FAMILIES[normalizedProviderKey(entry.provider.providerKey)]; + if (!providerFamily) return false; + const explicitSource = explicitModelSourceFamily(entry.model.modelId); + return !explicitSource || explicitSource === providerFamily; +} +function capabilitySignature(model: ModelsDevModel): string { + return JSON.stringify({ + attachment: model.attachment ?? null, + temperature: model.temperature ?? null, + toolCall: model.toolCall ?? null, + structuredOutput: model.structuredOutput ?? null, + modalities: { + input: [...model.modalities.input].sort(), + output: [...model.modalities.output].sort(), + }, + reasoning: model.reasoning, + reasoningOptions: model.reasoningOptions ?? null, + thinkingLevels: model.thinkingLevels, + }); +} + +function modelWithSharedCapabilities( + entries: readonly ModelsDevModel[], +): ModelsDevModel | undefined { + if (entries.length === 0) return undefined; + const signature = capabilitySignature(entries[0]); + if (entries.some((model) => capabilitySignature(model) !== signature)) return undefined; + return borrowedModel(entries); +} /** * Middle value of the numbers the publishers state. Even counts take the lower @@ -1415,12 +1419,7 @@ export class ModelsDevCatalog { for (const entry of this.lookupIndex.candidates(requested)) { (entry.provider === preferredProvider ? preferred : rest).push(entry); } - const candidates: Array<{ - model: ModelsDevModel; - provider: ModelsDevProvider; - score: number; - exact: boolean; - }> = []; + const candidates: Array<{ model: ModelsDevModel; provider: ModelsDevProvider; score: number }> = []; for (const { model, provider } of [...preferred, ...rest]) { // A known endpoint must not inherit another provider's capabilities. if (preferredProvider && provider !== preferredProvider) continue; @@ -1430,36 +1429,19 @@ export class ModelsDevCatalog { if (provider === preferredProvider) score += 100; if (apiMatches(input.baseUrl, provider.api)) score += 80; if (modelMatchesProvider(model, input.vendorKey)) score += 60; - candidates.push({ model, provider, score, exact }); + candidates.push({ model, provider, score }); } candidates.sort((left, right) => right.score - left.score || left.model.modelId.length - right.model.modelId.length, ); - /* - /* - A record the catalog publishes under exactly this id is this id's record. - Supported aliases — a release stamp, a thinking variant, a route leaf — - also reach a *shorter* sibling, and answering with that sibling would - attach another deployment's limits and capabilities to the row. So exact - candidates discard the alias-derived ones before the ambiguity test, which - is what keeps `foo-v2-0731` on its own record while a catalog that only - publishes `foo-v2` still answers it. - */ - const exactCandidates = candidates.filter((candidate) => candidate.exact); - const pool = exactCandidates.length > 0 ? exactCandidates : candidates; - /* Within the row's own catalog provider this is an exact-provider lookup: - the provider identity is known, so its record for the id — a direct hit - or a supported alias — is authoritative. - - With no provider identity the scores prove nothing about identity, so a - catalog answer is only usable when it is unambiguous — and only over the - alias-derived pool, since an exact record already outranks its aliases. - Two providers publishing the same id must not have one chosen for the - other. */ - const resolved = pool[0]?.model; - const result = preferredProvider - ? resolved - : pool.length === 1 ? resolved : undefined; + // 0 → unmatched; 1 → enrich; ≥2 prefer unique official/source provider + // agreeing with model source; else shared identical capabilities; else unmatched. + const official = candidates.filter(isOfficialSourceProvider); + const result = candidates.length === 1 + ? candidates[0].model + : official.length === 1 + ? official[0].model + : modelWithSharedCapabilities(candidates.map(({ model }) => model)); /* The row's own catalog provider did not publish this id. Borrowing needs a known provider identity to anchor on: the row resolved to a catalog provider whose own records are authoritative, so anything missing from it @@ -1467,40 +1449,13 @@ export class ModelsDevCatalog { dropping the model to the generic shape. This only fills a miss the lookup already had — a record the preferred - provider does publish stays authoritative. - - With no provider identity at all the scores still prove nothing about - identity, so no single publisher's record is adopted. What the publishers - of the *same model in another spelling* state is a different question, and - its answer can be claimed without adopting any one deployment's claims: - the publishers this app ships answer first, tool support follows the ones - that state it, capabilities are under-claimed and the limits are the lower - median. Two routes that merely share a leaf (`provider-a/foo` vs - `gateway/foo`) never enter that set, so an id whose identity is genuinely - unknown still resolves to nothing. */ - let borrowed = result ?? - (preferredProvider - ? this.borrowedAcrossProviders(input) - : unanchoredConsensus( - borrowPool( - candidates.filter((candidate) => sameModelSpelling(candidate.model.modelId, requested)), - requested, - ), - )); - /* - Nothing the catalog publishes answered for the id as served. A deployment - can put the model behind a route prefix or append a marker of its own - (`test/mimo-v2.5`, `mimo-v2.5-pro-test`), and the model is still the one - the catalog publishes. Reading that whole published id is the last resort, - so a suffix the catalog uses for models of its own — `-asr`, `-tts`, - `-voiceclone` — cannot turn an unknown id into a different model's record. - */ - if (!borrowed) { - for (const fallbackId of fallbackLookupIds(requested)) { - borrowed = this.borrowedAcrossProviders(input, fallbackId); - if (borrowed) break; - } - } + provider does publish stays authoritative. Without that anchor the scores + prove nothing about identity, and an unknown endpoint keeps the existing + behaviour: a unique unambiguous match, official disambiguation, shared + capabilities, or nothing. Deployment-marker / variant-suffix fallbacks + are intentionally not applied here (approved #1047 matching rules). */ + const borrowed = result ?? + (preferredProvider ? this.borrowedAcrossProviders(input) : undefined); // Cache the result (a miss included) so a repeated miss is also O(1) and // cannot grow the candidate index with query-dependent keys. this.lookupMemo.set(memoKey, borrowed); diff --git a/apps/desktop/electron/main/plugin-mcp.ts b/apps/desktop/electron/main/plugin-mcp.ts index 66d543ea1d..7a482184ad 100644 --- a/apps/desktop/electron/main/plugin-mcp.ts +++ b/apps/desktop/electron/main/plugin-mcp.ts @@ -173,6 +173,7 @@ function createStdioTransport( cwd: options.rootPath, env: launch.env, stdio: ["pipe", "pipe", "pipe"], + detached: process.platform !== "win32", // Arguments stay literal. Known launchers rewrite to a PE binary; remaining // Windows `.cmd` shims go through `cmd.exe /d /s /c` with quoted args. shell: false, @@ -183,14 +184,36 @@ function createStdioTransport( let closed = false; let buffer = ""; let lastStderr = ""; + let forceKillTimer: ReturnType | undefined; + const signalChildTree = (signal: NodeJS.Signals = "SIGTERM") => { + if (process.platform !== "win32" && child.pid) { + try { + process.kill(-child.pid, signal); + return; + } catch { + // The process group may already be gone; fall back to the direct child. + } + } + child.kill(signal); + }; + + const stopChild = () => { + closed = true; + child.stdin?.destroy(); + child.stdout?.destroy(); + child.stderr?.destroy(); + signalChildTree(); + forceKillTimer ??= setTimeout(() => signalChildTree("SIGKILL"), 1_000); + forceKillTimer.unref?.(); + }; child.stdout?.setEncoding("utf8"); child.stdout?.on("data", (chunk: string) => { buffer += chunk; if (buffer.length > MAX_STDIO_LINE_BYTES) { buffer = ""; handlers.onClose("mcp server sent an oversized message"); - child.kill(); + stopChild(); return; } let index = buffer.indexOf("\n"); @@ -240,10 +263,7 @@ function createStdioTransport( } child.stdin.write(`${JSON.stringify(message)}\n`); }, - close: () => { - closed = true; - child.kill(); - }, + close: stopChild, }; } diff --git a/apps/desktop/src/features/chat/composer/model.ts b/apps/desktop/src/features/chat/composer/model.ts index 718110e4d3..a38f71e407 100644 --- a/apps/desktop/src/features/chat/composer/model.ts +++ b/apps/desktop/src/features/chat/composer/model.ts @@ -127,14 +127,28 @@ export function thinkingProviderForModel( ): ProviderPublic | null | undefined { if (!provider || !modelId) return provider; const model = modelCatalog?.find((candidate) => sameComposerModelId(candidate.modelId, modelId)); - if (!model) return provider; const binding = provider.models.find((candidate) => - sameComposerModelId(candidate.id, model.modelId), + sameComposerModelId(candidate.id, modelId), ); const configuredLevels = binding ? THINKING_LEVELS.filter((level) => binding.thinkingLevels.includes(level)) : undefined; + + if (!model) { + // No catalog match: all thinking levels selectable, default off. + // A binding override still takes precedence when present. + const supportsReasoning = configuredLevels + ? configuredLevels.some((level) => level !== "off") + : true; + return { + ...provider, + supportsReasoning, + supportedThinkingLevels: + configuredLevels ?? [...THINKING_LEVELS], + }; + } + const supportsReasoning = configuredLevels ? configuredLevels.some((level) => level !== "off") : model.reasoning === true || model.capabilities.includes("reasoning"); diff --git a/apps/desktop/src/lib/composer-models.ts b/apps/desktop/src/lib/composer-models.ts index 8f1289ed07..44bb8bdc24 100644 --- a/apps/desktop/src/lib/composer-models.ts +++ b/apps/desktop/src/lib/composer-models.ts @@ -43,7 +43,7 @@ export function composerModelsForProvider( const metadata = (discovered ?? []).find((model) => sameComposerModelId(model.modelId, modelId), ); - const displayName = metadata?.displayName?.trim() || modelId; + const displayName = modelId; return metadata ? { ...metadata, modelId, displayName, providerId: provider.id } : { @@ -56,7 +56,7 @@ export function composerModelsForProvider( }); } -/** The Composer uses the configured alias, then published name, then wire id. */ +/** The Composer uses the configured alias when set, otherwise the wire id. */ export function composerModelDisplayName( provider: ConfiguredProvider | undefined, modelId: string, diff --git a/apps/desktop/test/composer-model-thinking-menu.test.mjs b/apps/desktop/test/composer-model-thinking-menu.test.mjs index 53d96fa0a3..022f128380 100644 --- a/apps/desktop/test/composer-model-thinking-menu.test.mjs +++ b/apps/desktop/test/composer-model-thinking-menu.test.mjs @@ -153,7 +153,7 @@ test("Composer uses alias labels while preserving the exact selected wire id", a test("reasoning projection uses the selected exact catalog row and binding", async () => { const source = await readComposerModule("model.ts"); assert.match(source, /sameComposerModelId\(candidate\.modelId, modelId\)/); - assert.match(source, /sameComposerModelId\(candidate\.id, model\.modelId\)/); + assert.match(source, /sameComposerModelId\(candidate\.id, modelId\)/); }); test("a model row spends the panel's width instead of stacking at its left edge", () => { diff --git a/apps/desktop/test/composer-models.test.mjs b/apps/desktop/test/composer-models.test.mjs index 757c0f495e..1d55fcc411 100644 --- a/apps/desktop/test/composer-models.test.mjs +++ b/apps/desktop/test/composer-models.test.mjs @@ -42,8 +42,8 @@ test("Composer only lists models configured for the provider", () => { assert.deepEqual( models.map(({ modelId, displayName }) => ({ modelId, displayName })), [ - { modelId: "claude-opus-4-6", displayName: "Claude Opus 4.6" }, - { modelId: "x-ai/grok-4.6", displayName: "Grok 4.6" }, + { modelId: "claude-opus-4-6", displayName: "claude-opus-4-6" }, + { modelId: "x-ai/grok-4.6", displayName: "x-ai/grok-4.6" }, ], ); }); @@ -78,10 +78,10 @@ test("legacy providers fall back to their default model binding", () => { ); assert.deepEqual(models.map((item) => item.modelId), ["legacy-model"]); - assert.equal(models[0].displayName, "Legacy model"); + assert.equal(models[0].displayName, "legacy-model"); }); -test("a configured alias labels its row without losing the published name", () => { +test("a configured alias labels its row while displayName shows the wire id", () => { const provider = { id: "deepseek", models: [{ ...binding("deepseek-v4-pro"), alias: " pro " }], @@ -92,10 +92,10 @@ test("a configured alias labels its row without losing the published name", () = ); assert.equal(models[0].modelId, "deepseek-v4-pro"); - assert.equal(models[0].displayName, "DeepSeek V4 Pro"); + assert.equal(models[0].displayName, "deepseek-v4-pro"); assert.equal(composerModelDisplayName(provider, "deepseek-v4-pro", models[0].displayName), "pro"); assert.equal(composerModelMatchesQuery(models[0], "provider", "PRO", "pro"), true); - assert.equal(composerModelMatchesQuery(models[0], "provider", "DeepSeek V4 Pro", "pro"), true); + assert.equal(composerModelMatchesQuery(models[0], "provider", "DeepSeek V4 Pro", "pro"), false); }); test("a configured alias is visible before discovery data is available", () => { const provider = { @@ -140,7 +140,7 @@ test("an exact binding alias wins over a broader equivalent id match", () => { ); }); -test("a blank alias leaves the published display name alone", () => { +test("a blank alias leaves the wire id as display name", () => { const models = composerModelsForProvider( { id: "deepseek", @@ -149,14 +149,14 @@ test("a blank alias leaves the published display name alone", () => { [model("deepseek-v4-pro", "DeepSeek V4 Pro")], ); - assert.equal(models[0].displayName, "DeepSeek V4 Pro"); + assert.equal(models[0].displayName, "deepseek-v4-pro"); assert.equal( composerModelDisplayName( { id: "deepseek", models: [{ ...binding("deepseek-v4-pro"), alias: " " }] }, "deepseek-v4-pro", - "DeepSeek V4 Pro", + "deepseek-v4-pro", ), - "DeepSeek V4 Pro", + "deepseek-v4-pro", ); }); @@ -248,8 +248,8 @@ test("prefixed and unprefixed wire ids remain separate even with one catalog nam { ...model("proxy/model", "Friendly"), capabilities: ["text", "vision"] }, ]); assert.deepEqual(rows.map(({ modelId, displayName }) => [modelId, displayName]), [ - ["proxy/model", "Friendly"], - ["model", "Friendly"], + ["proxy/model", "proxy/model"], + ["model", "model"], ]); assert.equal(composerModelDisplayName(provider, "proxy/model", rows[0].displayName), "Short"); assert.equal(composerModelMatchesQuery(rows[0], "relay", "Short", "Short"), true); diff --git a/apps/desktop/test/models-dev-catalog.test.mjs b/apps/desktop/test/models-dev-catalog.test.mjs index f4e62986c7..0b721f0f82 100644 --- a/apps/desktop/test/models-dev-catalog.test.mjs +++ b/apps/desktop/test/models-dev-catalog.test.mjs @@ -216,12 +216,12 @@ test("an unknown endpoint borrows only what every publisher of the id agrees on" } }); -test("an unknown endpoint can use a unique supported proxy alias", async (t) => { +test("an unknown endpoint can use a unique exact last-segment match", async (t) => { const catalog = await loadFixtureCatalog(t, { alpha: { models: { "shared-model": { id: "shared-model", reasoning: true } } }, }); assert.equal(catalog.findModel({ - vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId: "proxy/shared-model-thinking", + vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId: "proxy/shared-model", })?.providerKey, "alpha"); }); @@ -337,111 +337,170 @@ test("model IDs match provider namespaces without matching model variants", () = assert.equal(modelIdsMatch("proxy/openai/gpt-4o", "openai/gpt-4o"), true); assert.equal(modelIdsMatch("openai/gpt-4o", "proxy/gpt-4o"), false); }); -test("catalog metadata IDs match exact proxy paths and supported variants", () => { +test("catalog metadata IDs match only exact case-insensitive last segments", () => { for (const [catalogId, request] of [ ["claude-opus-4.6", "proxy/claude-opus-4.6"], - ["claude-opus-4.6", "custom/claude-opus-4.6"], - ["claude-opus-4.6", "relay/claude-opus-4.6"], - ["claude-opus-4.6", "gateway-01/claude-opus-4.6"], - ["openai/gpt-4o", "hub/openai/gpt-4o"], - ["claude-opus-4.6", "proxy/claude-opus-4.6-thinking"], - ["claude-opus-4.6", "claude-opus-4.6:thinking"], - ["claude-opus-4.6", "proxy/claude-opus-4.6-agent"], - ["claude-opus-4.6", "proxy/claude-opus-4.6-latest"], - ["claude-opus-4.6", "proxy/claude-opus-4.6-agent-thinking"], - ["proxy/nested/claude-opus-4.6@us-east", "claude-opus-4.6-thinking"], - ["anthropic-claude-opus-4.6", "claude-opus-4.6@us-east"], - ["google/gemini-pro-latest", "gemini-pro"], + ["openai/gpt-4o", "hub/gpt-4o"], + ["google/gemini-2.5-flash", "openai/gemini-2.5-flash"], + ["proxy/nested/claude-opus-4.6", "other/CLAUDE-OPUS-4.6"], + ["gateway-a/foo", "gateway-b/foo"], ]) { assert.equal(catalogModelIdsMatch(catalogId, request), true, `${catalogId} / ${request}`); assert.equal(catalogModelIdsMatch(request, catalogId), true, `${request} / ${catalogId}`); } for (const [catalogId, request] of [ - ["google/gemini-2.5-flash", "openai/gemini-2.5-flash"], - ["claude-opus-4.6", "myproxy-claude-opus-4.6"], - ["claude-opus-4.6", "myproxy-claude-opus-4.6-thinking"], - ["google/gemini-2.5-flash", "gemini-2.5-flash-high"], - ["google/gemini-2.5-flash", "gemini-2.5-flash-low"], - ["google/gemini-2.5-flash", "gemini-2.5-flash:minimal"], - ["claude-opus-5", "claude-opus-5-fast"], - ["gpt-4o", "gpt-4o-mini"], - ["gpt-4", "gpt-4o"], - ["model", "other-model"], - ["custom", "gemini-3.1-pro-preview-customtools"], - ["groq/whisper-large-v3", "deepseek-v3"], - ["vercel/bfl/flux-kontext-max", "qwen-max"], - ["alibaba/qwen3-asr-flash", "qwen-flash"], - ["openai/gpt-4o", "proxy/gpt-4o"], - ["google/gemini-2.5-flash", "proxy/gemini-2.5-flash"], - ["proxy/nested/claude-opus-4.6", "other/claude-opus-4.6"], - ["claude-opus-4-6-max", "qwen-max"], + ["claude-opus-4.6", "proxy/claude-opus-4.6-thinking"], + ["claude-opus-4.6", "claude-opus-4.6:thinking"], + ["claude-opus-4.6", "proxy/claude-opus-4.6-agent"], + ["claude-opus-4.6", "proxy/claude-opus-4.6-latest"], + ["anthropic-claude-opus-4.6", "claude-opus-4.6"], + ["google/gemini-pro-latest", "ag/gemini-pro-agent"], + ["foo", "foo-think"], + ["foo", "foo-agent"], + ["foo", "foo-latest"], + ["foo@us-east", "foo"], ]) { assert.equal(catalogModelIdsMatch(catalogId, request), false, `${catalogId} / ${request}`); assert.equal(catalogModelIdsMatch(request, catalogId), false, `${request} / ${catalogId}`); } - assert.equal(catalogModelIdsMatch("gateway-a/foo", "gateway-b/foo"), false); - assert.equal(catalogModelIdsMatch("foo", "foo-think"), true); - assert.equal(catalogModelIdsMatch("foo", "foo-agent"), true); - assert.equal(catalogModelIdsMatch("foo", "foo-latest"), true); }); -test("catalog lookup indexes exact path leaves and known vendor variants without broad aliases", async (t) => { - const ids = [ - "model", "custom", "groq/whisper-large-v3", "vercel/bfl/flux-kontext-max", - "alibaba/qwen3-asr-flash", "proxy/nested/claude-opus-4.6@us-east", - "anthropic-claude-sonnet-4", "openai.gpt-4o", "google/gemini-2.5-flash", - ]; +test("catalog lookup uses exact last segments and rejects ambiguous matches", async (t) => { const catalog = await loadFixtureCatalog(t, { - gateway: { models: Object.fromEntries(ids.map((id) => [id, { id }])) }, + gateway: { models: { + "model": { id: "model" }, + "nested/unique-model": { id: "nested/unique-model", reasoning: true }, + "google/gemini-pro-latest": { id: "google/gemini-pro-latest", reasoning: true }, + "route-a/shared-leaf": { id: "route-a/shared-leaf", reasoning: true }, + "route-b/shared-leaf": { id: "route-b/shared-leaf", reasoning: false }, + "regional@us-east": { id: "regional@us-east" }, + } }, }); - for (const [request, expected] of [ - ["other-model", undefined], - ["gemini-3.1-pro-preview-customtools", undefined], - ["deepseek-v3", undefined], - ["qwen-max", undefined], - ["qwen-flash", undefined], - ["myproxy-claude-opus-4.6", undefined], - ["gemini-2.5-flash-high", undefined], - ["gemini-2.5-flash-low", undefined], - ["claude-opus-4.6-thinking", "proxy/nested/claude-opus-4.6@us-east"], - ["proxy/claude-opus-4.6-thinking", undefined], - ["claude-sonnet-4-agent", "anthropic-claude-sonnet-4"], - ["proxy/gemini-2.5-flash", undefined], - ["gpt-4o@eu", "openai.gpt-4o"], - ]) { - assert.equal(catalog.findModel({ vendorKey: "custom", modelId: request })?.modelId, expected, request); - } - assert.equal(catalog.findModel({ vendorKey: "custom", modelId: "openai/gemini-2.5-flash" }), undefined); + + assert.equal( + catalog.findModel({ vendorKey: "custom", modelId: "proxy/UNIQUE-MODEL" })?.modelId, + "nested/unique-model", + ); + assert.equal(catalog.findModel({ vendorKey: "custom", modelId: "other/model" })?.modelId, "model"); + assert.equal(catalog.findModel({ vendorKey: "custom", modelId: "proxy/shared-leaf" }), undefined); + assert.equal(catalog.findModel({ vendorKey: "gateway", modelId: "proxy/shared-leaf" }), undefined); + assert.equal(catalog.findModel({ vendorKey: "custom", modelId: "ag/gemini-pro-agent" }), undefined); + assert.equal(catalog.findModel({ vendorKey: "custom", modelId: "regional" }), undefined); + assert.equal( + catalog.findModel({ vendorKey: "custom", modelId: "proxy/regional@us-east" })?.modelId, + "regional@us-east", + ); }); -test("matches supported proxy path and reasoning variants in catalog lookup", async (t) => { - const catalog = await loadFixtureCatalog(t); - for (const modelId of [ - "proxy/claude-opus-4.6", "proxy/claude-opus-4.6-thinking", - "proxy/claude-opus-4.6-agent", "claude-opus-4.6:thinking", - "anthropic-claude-opus-4.6", - ]) { - const match = catalog.findModel({ vendorKey: "custom", modelId }); - assert.equal(match?.modelId, "claude-opus-4.6", modelId); - assert.equal(match.reasoning, true); - } - for (const modelId of ["myproxy-claude-opus-4.6", "myproxy-claude-opus-4.6-thinking"]) { - assert.equal(catalog.findModel({ vendorKey: "custom", modelId }), undefined); +test("ambiguous leaf matches prefer one official provider", async (t) => { + for (const order of [["relay", "openai"], ["openai", "relay"]]) { + const definitions = { + relay: { models: { "route/shared-leaf": { + id: "route/shared-leaf", reasoning: false, tool_call: false, + limit: { context: 32_000 }, + } } }, + openai: { models: { "shared-leaf": { + id: "shared-leaf", reasoning: true, tool_call: true, + reasoning_options: [{ type: "effort", values: ["low", "medium", "high"] }], + limit: { context: 128_000 }, + } } }, + }; + const catalog = await loadFixtureCatalog(t, Object.fromEntries( + order.map((key) => [key, definitions[key]]), + )); + const match = catalog.findModel({ vendorKey: "custom", modelId: "proxy/shared-leaf" }); + assert.equal(match?.providerKey, "openai"); + assert.equal(match?.reasoning, true); + assert.equal(match?.limit.context, 128_000); } }); -test("metadata lookup can share routed leaves and narrow suffix aliases without merging bindings", async (t) => { +test("official source keys ignore a cloud publisher carrying another vendor's model", async (t) => { const catalog = await loadFixtureCatalog(t, { - gateway: { models: { - "gateway-a/foo": { id: "gateway-a/foo" }, - "bar-agent": { id: "bar-agent" }, + xai: { models: { "grok-4.6": { + id: "grok-4.6", reasoning: true, tool_call: true, + reasoning_options: [{ type: "effort", values: ["low", "medium", "high", "xhigh"] }], + limit: { context: 500_000 }, + } } }, + "google-vertex": { models: { "xai/grok-4.6": { + id: "xai/grok-4.6", reasoning: true, tool_call: true, + limit: { context: 524_288 }, + } } }, + relay: { models: { "x-ai/grok-4.6": { + id: "x-ai/grok-4.6", reasoning: false, tool_call: false, + } } }, + }); + const match = catalog.findModel({ vendorKey: "custom", modelId: "ycj/grok-4.6" }); + assert.equal(match?.providerKey, "xai"); + assert.equal(match?.limit.context, 500_000); +}); + +test("ambiguous non-official leaf matches share unanimous capabilities", async (t) => { + const catalog = await loadFixtureCatalog(t, { + relayA: { models: { "route-a/shared-leaf": { + id: "route-a/shared-leaf", reasoning: true, tool_call: true, + reasoning_options: [{ type: "effort", values: ["low", "high"] }], + modalities: { input: ["text", "image"], output: ["text"] }, + limit: { context: 64_000, output: 8_192 }, + } } }, + relayB: { models: { "route-b/shared-leaf": { + id: "route-b/shared-leaf", reasoning: true, tool_call: true, + reasoning_options: [{ type: "effort", values: ["low", "high"] }], + modalities: { input: ["image", "text"], output: ["text"] }, + limit: { context: 128_000, output: 16_384 }, + } } }, + }); + const match = catalog.findModel({ vendorKey: "custom", modelId: "proxy/shared-leaf" }); + assert.ok(match); + assert.equal(match.reasoning, true); + assert.equal(match.toolCall, true); + assert.deepEqual(match.thinkingLevels, ["low", "high"]); + assert.deepEqual(match.modalities.input.toSorted(), ["image", "text"]); +}); + +test("ambiguous leaf matches with conflicting capabilities remain unmatched", async (t) => { + const catalog = await loadFixtureCatalog(t, { + relayA: { models: { "route-a/shared-leaf": { + id: "route-a/shared-leaf", reasoning: true, tool_call: true, + } } }, + relayB: { models: { "route-b/shared-leaf": { + id: "route-b/shared-leaf", reasoning: false, tool_call: true, + } } }, + }); + assert.equal( + catalog.findModel({ vendorKey: "custom", modelId: "proxy/shared-leaf" }), + undefined, + ); +}); + +test("gemini agent routes do not collapse to gemini latest catalog entries", async (t) => { + const catalog = await loadFixtureCatalog(t, { + google: { models: { + "gemini-pro-latest": { id: "gemini-pro-latest", reasoning: true }, } }, }); - assert.equal(modelIdsMatch("gateway-a/foo", "gateway-b/foo"), false); - assert.equal(modelIdsMatch("bar", "bar-agent"), false); - assert.equal(catalog.findModel({ modelId: "gateway-b/foo" }), undefined); - assert.equal(catalog.findModel({ modelId: "bar-thinking" })?.modelId, "bar-agent"); + assert.equal( + catalog.findModel({ vendorKey: "custom", modelId: "ag/gemini-pro-agent" }), + undefined, + ); +}); + +test("catalog lookup never collapses reasoning or release suffix variants", async (t) => { + const catalog = await loadFixtureCatalog(t); + const exact = catalog.findModel({ vendorKey: "custom", modelId: "proxy/claude-opus-4.6" }); + assert.equal(exact?.modelId, "claude-opus-4.6"); + assert.equal(exact.reasoning, true); + + for (const modelId of [ + "proxy/claude-opus-4.6-thinking", + "proxy/claude-opus-4.6-agent", + "proxy/claude-opus-4.6-latest", + "claude-opus-4.6:thinking", + "anthropic-claude-opus-4.6", + ]) { + assert.equal(catalog.findModel({ vendorKey: "custom", modelId }), undefined, modelId); + } }); /* @@ -764,7 +823,7 @@ test("an exact record outranks a shipped publisher's other spelling of the id", assert.ok(info.capabilities.includes("audio")); }); -test("borrowing stays exact-id, provider-scoped and absent for unknown ids", async (t) => { +test("unique leaf enrichment coexists with exact-id borrowing and unknown misses", async (t) => { const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, publisherA: { models: { @@ -773,10 +832,11 @@ test("borrowing stays exact-id, provider-scoped and absent for unknown ids", asy } }, }); const input = { vendorKey: "gateway", baseUrl: "https://gateway.example/v1" }; - // A near-miss id is not a reason to reuse a sibling's limits. + // Near-miss and unknown ids do not reuse a sibling's limits. assert.equal(catalog.findModel({ ...input, modelId: "Vendor/Known-3000" }), undefined); - assert.equal(catalog.findModel({ ...input, modelId: "Other/Known" }), undefined); assert.equal(catalog.findModel({ ...input, modelId: "unknown-model-xyz" }), undefined); + // A different route with the same unique final segment is valid enrichment. + assert.equal(catalog.findModel({ ...input, modelId: "Other/Known" })?.modelId, "Vendor/Known"); // Exact ids still resolve, including a case-only difference. assert.equal(catalog.findModel({ ...input, modelId: "vendor/known" })?.limit.context, 262_144); assert.equal(catalog.findModel({ ...input, modelId: "Vendor/Known" })?.limit.context, 262_144); @@ -1580,7 +1640,7 @@ const customGateway = { baseUrl: "https://gateway.example/v1", }; -test("a custom Anthropic gateway reads Anthropic's own record and its thinking shape", async (t) => { +test("a custom gateway enriches an ambiguous Claude leaf from the official provider", async (t) => { const catalog = await loadFixtureCatalog(t, ambiguousClaudeFixture); const config = catalogModelConfigFor(catalog, { ...customGateway, @@ -1588,20 +1648,10 @@ test("a custom Anthropic gateway reads Anthropic's own record and its thinking s modelId: "claude-opus-5-5", }); - /* - Anthropic publishes this id, so the record of the publisher the app ships - answers — not a median shared with a reseller's smaller deployment of the - same id, which would halve the window Anthropic itself states. - */ assert.equal(config.source, "models.dev"); assert.equal(config.contextWindow, 1_000_000); assert.equal(config.maxTokens, 128_000); assert.equal(config.reasoning, true); - /* - No publisher describes this endpoint's reasoning wire shape, so the borrowed - record states none — and Anthropic's own record is what says whether the id - takes adaptive or budget thinking. - */ assert.deepEqual(config.reasoningOptions, [ { type: "effort", values: ["low", "medium", "high", "xhigh", "max"] }, ]); @@ -1615,15 +1665,18 @@ test("a custom Anthropic gateway reads Anthropic's own record and its thinking s }); }); -test("the Anthropic thinking fallback stays off other wire APIs and non-Claude ids", async (t) => { +test("official-provider enrichment is API-style independent while unknown ids stay generic", async (t) => { const catalog = await loadFixtureCatalog(t, ambiguousClaudeFixture); const completions = catalogModelConfigFor(catalog, { ...customGateway, apiStyle: "chat_completions", modelId: "claude-opus-5-5", }); - assert.equal(completions.reasoningOptions, undefined); - assert.equal(completions.thinkingLevelMap, undefined); + assert.equal(completions.source, "models.dev"); + assert.equal(completions.contextWindow, 1_000_000); + assert.deepEqual(completions.reasoningOptions, [ + { type: "effort", values: ["low", "medium", "high", "xhigh", "max"] }, + ]); // An id absent from every publisher stays generic, even on Anthropic Messages. const unlisted = catalogModelConfigFor(catalog, { diff --git a/apps/desktop/test/plugin-mcp.test.mjs b/apps/desktop/test/plugin-mcp.test.mjs index 9ffc469269..18c255200c 100644 --- a/apps/desktop/test/plugin-mcp.test.mjs +++ b/apps/desktop/test/plugin-mcp.test.mjs @@ -244,21 +244,34 @@ test("a stdio server that cannot start fails the handshake, not the process", as assert.match(String(failure.message), /exited with code/); }); -test("a slow server times out instead of hanging the load", async (t) => { +test("a slow server times out instead of hanging the load", async () => { const dir = stdioPlugin(); - writeFileSync(join(dir, "server.mjs"), "setInterval(() => {}, 1000);\n"); + const pidFile = join(dir, "pid"); + writeFileSync( + join(dir, "server.mjs"), + 'import { writeFileSync } from "node:fs";\nwriteFileSync(process.env.STUB_PID_FILE, String(process.pid));\nsetInterval(() => {}, 1000);\n', + ); const client = new McpServerClient({ pluginId: "com.example.mcp", rootPath: dir, server: { id: "stub", transport: "stdio", command: "node", args: ["./server.mjs"] }, - values: {}, + values: { STUB_PID_FILE: pidFile }, connectTimeoutMs: 250, }); - t.after(() => client.close()); await assert.rejects(client.connect(), (error) => { assert.equal(error.code, "TIMEOUT"); return true; }); + const pid = Number(readFileSync(pidFile, "utf8")); + for (let attempt = 0; attempt < 40; attempt += 1) { + try { + process.kill(pid, 0); + } catch { + return; + } + await new Promise((r) => setTimeout(r, 50)); + } + assert.fail("timed-out stdio mcp child survived handshake cleanup"); }); /** Streamable-HTTP stub: JSON for the handshake, SSE for discovery. */ diff --git a/packages/shared/src/thinking-levels.test.ts b/packages/shared/src/thinking-levels.test.ts index a81ebc0d97..a313ba3105 100644 --- a/packages/shared/src/thinking-levels.test.ts +++ b/packages/shared/src/thinking-levels.test.ts @@ -54,14 +54,14 @@ describe("initialThinkingLevelForBinding", () => { ).toBe("high"); }); - it("falls back to the strongest enabled level when no default is stored", () => { + it("defaults to off when no default is stored", () => { expect( initialThinkingLevelForBinding({ thinkingLevels: ["low", "high", "max"], defaultThinkingLevel: null, }), - ).toBe("max"); - expect(initialThinkingLevelForBinding(undefined, ["low", "high"])).toBe("high"); + ).toBe("off"); + expect(initialThinkingLevelForBinding(undefined, ["low", "high"])).toBe("off"); }); it("honors an explicit off default and empty bindings", () => { diff --git a/packages/shared/src/thinking-levels.ts b/packages/shared/src/thinking-levels.ts index 9336b727a5..7efac537b4 100644 --- a/packages/shared/src/thinking-levels.ts +++ b/packages/shared/src/thinking-levels.ts @@ -91,11 +91,10 @@ export function nearestSupportedThinkingLevel( /** * Thinking level a new draft or session starts at for a model binding. * - * Prefer the stored default when it is still enabled, including `omit` on a - * reasoning binding. Otherwise clamp that default onto the enabled ladder. - * With no stored default, fall back to the strongest enabled level so a - * reasoning model never starts at `off` merely because Settings has not - * picked a default yet. + * Prefer an explicitly stored default when it is still enabled, including + * `omit` on a reasoning binding. Otherwise clamp that explicit default onto + * the enabled ladder. With no stored default, start at `off`; available + * reasoning levels remain selectable but are never enabled implicitly. */ export function initialThinkingLevelForBinding( binding: ThinkingLevelBindingSource | null | undefined, @@ -105,7 +104,7 @@ export function initialThinkingLevelForBinding( const stored = binding?.defaultThinkingLevel; if (stored === "omit") return enablesReasoning(enabled) ? "omit" : "off"; if (stored != null) return nearestSupportedThinkingLevel(stored, enabled); - return highestSupportedThinkingLevel(enabled); + return "off"; } /** Published record a thinking-level candidate list can be derived from. */ diff --git a/packages/shared/src/types/models.ts b/packages/shared/src/types/models.ts index 2f98f97dce..b21f1e3534 100644 --- a/packages/shared/src/types/models.ts +++ b/packages/shared/src/types/models.ts @@ -150,22 +150,7 @@ function extractKnownVendor(id: string): string | undefined { return undefined; } -function pathLeaf(id: string): string { - const slash = id.lastIndexOf("/"); - return slash >= 0 ? id.slice(slash + 1) : id; -} - -function exactPathAliasMatch(left: string, right: string): boolean { - // Two independently routed paths cannot be identified by their leaf alone. - if (left.includes("/") === right.includes("/")) return false; - if (pathLeaf(left) !== pathLeaf(right)) return false; - - const leftVendor = extractKnownVendor(left); - const rightVendor = extractKnownVendor(right); - return !leftVendor || !rightVendor || leftVendor === rightVendor; -} - -function normalizedMatch(left: string, right: string, allowPathLeaf = false): boolean { +function normalizedMatch(left: string, right: string): boolean { const leftVendor = extractKnownVendor(left); const rightVendor = extractKnownVendor(right); if (leftVendor && rightVendor && leftVendor !== rightVendor) return false; @@ -179,7 +164,11 @@ function normalizedMatch(left: string, right: string, allowPathLeaf = false): bo } } - return allowPathLeaf && exactPathAliasMatch(left, right); + return false; +} +function pathLeaf(id: string): string { + const slash = id.lastIndexOf("/"); + return slash >= 0 ? id.slice(slash + 1) : id; } /** Configured-model identity: compare the complete wire ID, not catalog aliases. */ @@ -197,30 +186,20 @@ export function modelIdsMatch(candidate: string, requested: string): boolean { return normalizedMatch(stripRegion(left), stripRegion(right)); } -/** Broader metadata-only aliases; never use for configured binding identity. */ +/** + * Catalog enrichment matcher: take the **last `/`-segment** of each side, + * compare case-insensitively. The caller enforces uniqueness (exactly 1 + * catalog hit ⇒ enrichment; 0 or ≥2 ⇒ no match). + * + * This deliberately does **not** strip `-thinking`, `-agent`, `-latest`, + * vendor-dash prefixes, or any other fuzzy suffix. The old variant-suffix + * and vendor-prefix logic caused cross-model false positives. + */ export function catalogModelIdsMatch(candidate: string, requested: string): boolean { const left = candidate.trim().toLowerCase(); const right = requested.trim().toLowerCase(); if (!left || !right) return false; - - const cleanLeft = stripRegion(left); - const cleanRight = stripRegion(right); - // Exact id, known vendor prefix, route leaf, then thinking/agent/latest. - if (normalizedMatch(cleanLeft, cleanRight, true)) return true; - - const variantLeft = stripVariantSuffix(cleanLeft); - const variantRight = stripVariantSuffix(cleanRight); - if (normalizedMatch(variantLeft, variantRight, true)) return true; - - /* Published release stamps are tried last, so a dated snapshot can never - displace the exact id or a documented alias. A catalog that publishes both - `foo-v2` and `foo-v2-0731` therefore still answers `foo-v2-0731` with its - own record: the caller's exact-first ranking decides between them. */ - return normalizedMatch( - stripReleaseSuffix(variantLeft), - stripReleaseSuffix(variantRight), - true, - ); + return pathLeaf(left) === pathLeaf(right); } /** From 9069108c9782be9022622dcc7adfa9ed7478acf3 Mon Sep 17 00:00:00 2001 From: aurorax-neo <15047150695@163.com> Date: Sat, 26 Sep 2026 22:54:25 +0800 Subject: [PATCH 2/5] test: align catalog expectations with last-segment matching Update upstream tests that assumed release-stamp stripping, deployment marker fallbacks, or majority/consensus borrowing so they match the approved #1047 matcher and official/shared-capabilities disambiguation. --- apps/desktop/test/models-dev-catalog.test.mjs | 204 +++++------------- .../test/provider-endpoint-metadata.test.mjs | 59 +++-- packages/shared/src/model-identity.test.ts | 28 +-- 3 files changed, 95 insertions(+), 196 deletions(-) diff --git a/apps/desktop/test/models-dev-catalog.test.mjs b/apps/desktop/test/models-dev-catalog.test.mjs index 0b721f0f82..69ed8a8468 100644 --- a/apps/desktop/test/models-dev-catalog.test.mjs +++ b/apps/desktop/test/models-dev-catalog.test.mjs @@ -164,7 +164,7 @@ test("cached matches remain scoped to the requested provider and endpoint", asyn } }); -test("an unknown endpoint borrows only what every publisher of the id agrees on", async (t) => { +test("unknown endpoints do not inherit ambiguous cross-provider metadata", async (t) => { for (const order of [["alpha", "beta"], ["beta", "alpha"]]) { const fixture = Object.fromEntries(order.map((key) => [key, { api: `https://${key}.example/v1`, @@ -177,32 +177,11 @@ test("an unknown endpoint borrows only what every publisher of the id agrees on" }, }])); const catalog = await loadFixtureCatalog(t, fixture); - /* - No provider identity: neither publisher may answer alone. What every one of - them states can be claimed, and it under-claims — reasoning only when all - report it, limits as the lower median of the claims. - */ - const unknown = { vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId: "shared-model" }; - const consensus = catalog.findModel(unknown); - assert.equal(consensus?.limit.context, 32_000, "the lower median, not the larger claim"); - assert.equal(consensus?.limit.output, 8_192); - assert.equal(consensus?.reasoning, false, "one publisher's reasoning claim must not transfer"); - assert.equal(catalog.findModel(unknown), consensus, "a consensus is cached like any other answer"); - /* - A deployment that puts the model behind a route prefix still serves that - model: `proxy/shared-model` reads the record `shared-model` is published - under, with the same consensus rules. Two *different* route paths that - merely share a leaf are still not one model — the matcher rejects those - rather than averaging them. - */ - const prefixed = catalog.findModel({ - vendorKey: "custom", - baseUrl: "https://relay.example/v1", - modelId: "proxy/shared-model", - }); - assert.equal(prefixed?.modelId, "shared-model"); - assert.equal(prefixed?.limit.context, 32_000); - assert.equal(prefixed?.reasoning, false); + for (const modelId of ["shared-model", "proxy/shared-model"]) { + const unknown = { vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId }; + assert.equal(catalog.findModel(unknown), undefined, `ambiguous ${modelId} must miss`); + assert.equal(catalog.findModel(unknown), undefined, "ambiguous misses are cached"); + } const alpha = catalog.findModel({ vendorKey: "custom", baseUrl: "https://alpha.example/v1", modelId: "shared-model", }); @@ -252,17 +231,7 @@ test("a shared catalog API with no vendor key cannot select the first publisher" limit: { context: key === "alpha" ? 128_000 : 32_000, output: 8_192 } } }, }]))); assert.equal(catalog.providerKeyForRow({ vendorKey: "custom", baseUrl: "https://gateway.example/v1" }), undefined); - /* - The row has no publisher of its own, so the id answers with what both - publishers of that endpoint state — never with the first one indexed. - */ - const consensus = catalog.findModel({ - vendorKey: "custom", - baseUrl: "https://gateway.example/v1", - modelId: "shared-model", - }); - assert.equal(consensus?.reasoning, false); - assert.equal(consensus?.limit.context, 32_000); + assert.equal(catalog.findModel({ vendorKey: "custom", baseUrl: "https://gateway.example/v1", modelId: "shared-model" }), undefined); assert.equal(catalog.findModel({ vendorKey: "beta", baseUrl: "https://gateway.example/v1", modelId: "shared-model" })?.providerKey, "beta"); } }); @@ -578,31 +547,19 @@ test("a borrowed record under-claims instead of asserting one publisher's extras assert.equal(info.capabilities.includes("reasoning"), false); }); -test("an even split on tool support claims no tools but still borrows the limits", async (t) => { - /* - Tool support follows the majority of the publishers that state it. One - dissenting reseller must not decide the claim for a deployment it does not - describe, but neither may one agreeing reseller: an id a relay lists can be - stated by a hundred publishers. Two publishers that split evenly state no - majority, so the borrow keeps the limits and leaves tool support unclaimed - instead of refusing outright — refusing is what left whole model lists on - the generic seed. - */ +test("an even split on tool support leaves conflicting leaves unmatched", async (t) => { + // Conflicting capability signatures (tool_call) mean shared-capabilities enrichment + // must miss — no majority/consensus borrow for an unanchored / empty-gateway row. const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, publisherA: { models: { "Vendor/Split": { id: "Vendor/Split", tool_call: true, limit: { context: 262_144 } } } }, publisherB: { models: { "Vendor/Split": { id: "Vendor/Split", tool_call: false, limit: { context: 262_144 } } } }, }); - const match = catalog.findModel({ + assert.equal(catalog.findModel({ vendorKey: "gateway", baseUrl: "https://gateway.example/v1", modelId: "Vendor/Split", - }); - assert.ok(match, "a split must not drop the id to the generic shape"); - assert.equal(match.toolCall, undefined, "an even split claims no tool support"); - assert.equal(match.limit.context, 262_144); - const info = modelInfoFromModelsDev(match, "provider-1"); - assert.equal(info.capabilities.includes("tools"), false); + }), undefined); }); test("a borrowed record never replaces what the row's own catalog provider publishes", async (t) => { @@ -624,11 +581,7 @@ test("a borrowed record never replaces what the row's own catalog provider publi assert.equal(match.toolCall, false); }); -test("publishers that disagree about capabilities narrow the borrowed record", async (t) => { - // A disputed capability shape is never resolved by picking one publisher: - // the disagreement narrows the claim instead of voiding the borrow. - // Reasoning follows the intersection, so it is reported only when every - // publisher states it. +test("publishers that disagree about capabilities remain unmatched", async (t) => { const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, publisherA: { models: { "Vendor/Disputed": { @@ -638,21 +591,16 @@ test("publishers that disagree about capabilities narrow the borrowed record", a id: "Vendor/Disputed", tool_call: false, reasoning: false, limit: { context: 262_144 }, } } }, }); - const match = catalog.findModel({ + assert.equal(catalog.findModel({ vendorKey: "gateway", baseUrl: "https://gateway.example/v1", modelId: "Vendor/Disputed", - }); - assert.ok(match, "a disputed capability shape must not void the borrow"); - assert.equal(match.toolCall, undefined); - assert.equal(match.reasoning, false); - assert.equal(match.limit.context, 262_144); + }), undefined); }); -test("a majority of the publishers that state tool support decides the borrow", async (t) => { - // The counterpart of the even split above: with a clear majority the claim - // is stated, so a relay listing an id many publishers agree on keeps its - // tool support instead of dropping to text-only defaults. +test("a majority of publishers with conflicting capability signatures stays unmatched", async (t) => { + // #1047 does not majority-vote across publishers for an empty/unanchored row. + // Conflicting tool_call signatures ⇒ unmatched. const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, publisherA: { models: { "Vendor/Majority": { id: "Vendor/Majority", tool_call: true, limit: { context: 262_144 } } } }, @@ -660,26 +608,16 @@ test("a majority of the publishers that state tool support decides the borrow", publisherC: { models: { "Vendor/Majority": { id: "Vendor/Majority", tool_call: true, limit: { context: 1_048_576 } } } }, publisherD: { models: { "Vendor/Majority": { id: "Vendor/Majority", tool_call: false, limit: { context: 1_048_576 } } } }, }); - const match = catalog.findModel({ + assert.equal(catalog.findModel({ vendorKey: "gateway", baseUrl: "https://gateway.example/v1", modelId: "Vendor/Majority", - }); - assert.ok(match, "a clear majority still borrows"); - assert.equal(match.toolCall, true); - // Limits are still the lower median, so one small window is not rounded up - // and one dissenting publisher does not drag the majority's window down. - assert.equal(match.limit.context, 1_048_576); - const info = modelInfoFromModelsDev(match, "provider-1"); - assert.equal(info.capabilities.includes("tools"), true); + }), undefined); }); -test("the publishers this app ships decide an id that resellers also state", async (t) => { - /* - A relay's list names ids arbitrary resellers carry too, and a reseller's - flags describe its own deployment: here two resellers outvote the shipped - publisher and both are wrong about the tools it accepts. - */ +test("conflicting reseller and shipped records for a leaf stay unmatched without an official", async (t) => { + // deepseek is not an official/source family for #1047 disambiguation, and the + // capability signatures conflict, so an unanchored relay must miss. const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, deepseek: { models: { "shared-model": { @@ -692,16 +630,11 @@ test("the publishers this app ships decide an id that resellers also state", asy id: "shared-model", tool_call: false, limit: { context: 131_072, output: 8_192 }, } } }, }); - const match = catalog.findModel({ + assert.equal(catalog.findModel({ vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId: "shared-model", - }); - assert.ok(match, "a shipped publisher states the id"); - assert.equal(match.providerKey, "deepseek", "the shipped publisher's record answers"); - assert.equal(match.toolCall, true); - assert.equal(match.limit.context, 1_000_000); - assert.equal(match.limit.output, 393_216); + }), undefined); }); test("resellers still answer for an id no shipped publisher states", async (t) => { @@ -726,13 +659,9 @@ test("resellers still answer for an id no shipped publisher states", async (t) = assert.equal(match.limit.context, 262_144); }); -test("a route prefix and a deployment marker still reach the published model", async (t) => { - /* - Relays name a published model behind a route of their own and append markers - of their own: `test/mimo-v2.5`, `mimo-v2.5-thinking`, `test/mimo-v2.5-pro-test`. - All three are that published model. A suffix the catalog uses for a model of - its own — `-asr` — is not a marker, so an id carrying one stays unknown. - */ +test("a route prefix reaches a unique leaf while deployment markers stay unmatched", async (t) => { + // Exact last-segment match allows `test/mimo-v2.5` → `mimo-v2.5`. + // Markers like `-thinking` / `-test` are part of the leaf and do not strip. const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, xiaomi: { models: { @@ -749,18 +678,15 @@ test("a route prefix and a deployment marker still reach the published model", a const relay = (modelId) => catalog.findModel({ vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId }); - for (const served of ["test/mimo-v2.5", "mimo-v2.5-thinking", "mimo-v2.5"]) { + for (const served of ["test/mimo-v2.5", "mimo-v2.5"]) { assert.equal(relay(served)?.modelId, "mimo-v2.5", `${served} reads the published model`); assert.equal(relay(served)?.limit.context, 1_048_576); } const info = modelInfoFromModelsDev(relay("test/mimo-v2.5"), "provider-1"); assert.ok(info.capabilities.includes("vision")); - // The prefix and the marker together, as one deployment writes them. - assert.equal(relay("test/mimo-v2.5-pro-test")?.modelId, "mimo-v2.5-pro"); - assert.equal(relay("test/mimo-v2.5-pro-test")?.limit.output, 131_072); - - // `-asr` names a model of its own, so nothing is borrowed for it. + assert.equal(relay("mimo-v2.5-thinking"), undefined); + assert.equal(relay("test/mimo-v2.5-pro-test"), undefined); assert.equal(relay("mimo-v2.5-asr"), undefined); }); @@ -791,13 +717,9 @@ test("a shipped publisher answers first for a row anchored to a reseller", async assert.equal(match.limit.output, 500_000); }); -test("an exact record outranks a shipped publisher's other spelling of the id", async (t) => { - /* - Both are shipped publishers, and `XiaomiMiMo/MiMo-V2.5` is the same model as - `mimo-v2.5`. Only the exact id decides what the model can take: a publisher - that lists the text half alone must not narrow the model's own record into - text-only. - */ +test("ambiguous same-leaf spellings with conflicting modalities stay unmatched", async (t) => { + // `mimo-v2.5` and `XiaomiMiMo/MiMo-V2.5` share a case-insensitive leaf but + // disagree on modalities, and neither publisher is an official/source family. const catalog = await loadFixtureCatalog(t, { gateway: { api: "https://gateway.example/v1", models: {} }, xiaomi: { models: { "mimo-v2.5": { @@ -811,16 +733,11 @@ test("an exact record outranks a shipped publisher's other spelling of the id", limit: { context: 1_048_576, output: 131_072 }, } } }, }); - const match = catalog.findModel({ + assert.equal(catalog.findModel({ vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId: "mimo-v2.5", - }); - assert.equal(match?.providerKey, "xiaomi"); - assert.deepEqual(match.modalities.input, ["text", "image", "audio"]); - const info = modelInfoFromModelsDev(match, "provider-1"); - assert.ok(info.capabilities.includes("vision"), "the sibling's text-only record must not narrow it"); - assert.ok(info.capabilities.includes("audio")); + }), undefined); }); test("unique leaf enrichment coexists with exact-id borrowing and unknown misses", async (t) => { @@ -1717,7 +1634,7 @@ test("a resolved catalog record still wins over the Anthropic thinking fallback" under the id that owns the weights. The alias only ever borrows metadata: the id the row is addressed with stays the discovery result. */ -test("a dated snapshot borrows its published model's metadata without changing the wire id", async (t) => { +test("a dated snapshot leaf does not borrow an undated published model", async (t) => { const catalog = await loadFixtureCatalog(t, { mify: { api: "https://api.mify.example/v1", @@ -1732,17 +1649,13 @@ test("a dated snapshot borrows its published model's metadata without changing t }, }, }); - const model = catalog.findModel({ vendorKey: "mify", modelId: "mify/mimo-v2.5-pro-0731" }); - assert.equal(model?.modelId, "mimo-v2.5-pro"); - assert.equal(model?.limit.context, 262_144); - - const info = modelInfoFromModelsDev(model, "provider-1"); - const binding = bindingForCustomModelInfo("mify/mimo-v2.5-pro-0731", info); - assert.equal(binding.id, "mify/mimo-v2.5-pro-0731"); - assert.equal(binding.contextWindow, 262_144); + assert.equal( + catalog.findModel({ vendorKey: "mify", modelId: "mify/mimo-v2.5-pro-0731" }), + undefined, + ); }); -test("an exact record answers an id its shorter siblings would only alias", async (t) => { +test("exact published leaves resolve without release-stamp aliasing", async (t) => { const catalog = await loadFixtureCatalog(t, { alpha: { api: "https://alpha.example/v1", models: { "foo-v2": { id: "foo-v2", limit: { context: 32_000 } }, @@ -1756,22 +1669,19 @@ test("an exact record answers an id its shorter siblings would only alias", asyn assert.equal(exact?.modelId, "foo-v2-0731"); assert.equal(exact?.limit.context, 128_000); - // A stamp nothing publishes borrows the model it was cut from. - const borrowed = catalog.findModel({ vendorKey: "alpha", modelId: "foo-v2-0815" }); - assert.equal(borrowed?.modelId, "foo-v2"); + // Unpublished stamps do not strip down to a shorter sibling. + assert.equal(catalog.findModel({ vendorKey: "alpha", modelId: "foo-v2-0815" }), undefined); - /* - An id only its siblings state answers with their consensus rather than with - one publisher's record — the dated sibling's claim included, the lower - median of the three is what every publisher can be said to support. - */ - const consensus = catalog.findModel({ + // Same leaf, identical capabilities (limits are not part of the signature) ⇒ + // shared-capabilities enrichment with the lower median window. + const shared = catalog.findModel({ vendorKey: "custom", baseUrl: "https://relay.example/v1", modelId: "foo-v2", }); - assert.equal(consensus?.limit.context, 64_000); - // …but a record the catalog publishes in full answers for itself. + assert.equal(shared?.limit.context, 32_000); + + // A uniquely published dated leaf still enriches. const published = catalog.findModel({ vendorKey: "custom", baseUrl: "https://relay.example/v1", @@ -1818,7 +1728,7 @@ test("a custom endpoint on a uniquely published host inherits that publisher", a ); }); -test("a host two publishers share borrows only their consensus", async (t) => { +test("a host two publishers share leaves conflicting leaves unmatched", async (t) => { const catalog = await loadFixtureCatalog(t, { alpha: { api: "https://shared.example/api/paas/v4", @@ -1829,19 +1739,11 @@ test("a host two publishers share borrows only their consensus", async (t) => { models: { "glm-5.3": { id: "glm-5.3", reasoning: false, limit: { context: 32_000 } } }, }, }); - /* - The typed path is not published, so the host is the only evidence — and it - names two publishers, neither of which may answer alone. What both state is - what a row without an identity gets: reasoning only when both claim it, and - the lower median of the limits. - */ - const consensus = catalog.findModel({ + assert.equal(catalog.findModel({ vendorKey: "custom", baseUrl: "https://shared.example/v1", modelId: "glm-5.3", - }); - assert.equal(consensus?.reasoning, false); - assert.equal(consensus?.limit.context, 32_000); + }), undefined); // Naming the publisher still resolves it to that publisher's own record. const alpha = catalog.findModel({ vendorKey: "alpha", diff --git a/apps/desktop/test/provider-endpoint-metadata.test.mjs b/apps/desktop/test/provider-endpoint-metadata.test.mjs index 1f34c6e48a..78a5c2e035 100644 --- a/apps/desktop/test/provider-endpoint-metadata.test.mjs +++ b/apps/desktop/test/provider-endpoint-metadata.test.mjs @@ -165,14 +165,11 @@ test("a hand-typed id on the same relay answers with the same record", async (t) assert.equal(unknown.info, null); }); -test("a relay's list keeps the tool support a majority of publishers states", async (t) => { +test("a relay enriches unique/official leaves and leaves ambiguous or marker leaves unmatched", async (t) => { /* - Reported misses from a relay's own list. `deepseek-v4-flash` is stated by - dozens of publishers and `mimo-v2.5-pro` by many, a few of which say - `tool_call: false`, so requiring unanimity dropped both to the generic seed - even though the id is plainly published. Tool support now follows the - majority. A served id whose own record is an audio model resolves with it - too, while an id no publisher states still gets nothing. + #1047: exact last-segment match + official/shared-capabilities disambiguation. + Multi-publisher leaves with conflicting capabilities stay generic. Deployment + markers (`-1m`) stay part of the leaf and do not strip to a sibling. */ const row = rowOf({ id: "row-4", name: "Relay", baseUrl: "https://relay.example/v1" }); const { result } = await handlersFor(t, row, { @@ -181,49 +178,45 @@ test("a relay's list keeps the tool support a majority of publishers states", as { id: "mimo-v2.5-pro" }, { id: "mimo-v2.5-tts" }, { id: "gemini-2.5-pro-1m" }, + { id: "claude-sonnet-4-5" }, ], }); const byId = new Map(result.models.map((model) => [model.modelId, model])); - const deepseek = byId.get("deepseek-v4-flash"); - assert.equal(deepseek.catalogSource, "models.dev"); - assert.ok(deepseek.capabilities.includes("tools"), "a stated majority must be claimed"); - assert.equal(deepseek.contextWindow, 1_000_000); + // Ambiguous non-official leaves with conflicting publisher caps stay generic. + assert.equal(byId.get("deepseek-v4-flash").catalogSource, undefined); + assert.equal(byId.get("mimo-v2.5-pro").catalogSource, undefined); - const mimo = byId.get("mimo-v2.5-pro"); - assert.equal(mimo.catalogSource, "models.dev"); - assert.ok(mimo.capabilities.includes("tools")); - assert.equal(mimo.contextWindow, 1_048_576); - - // The listing decides which models a row *offers*; the lookup answers for the - // ids the service actually serves, audio-only ones included. + // Unique leaf still enriches. const tts = byId.get("mimo-v2.5-tts"); assert.equal(tts.catalogSource, "models.dev"); assert.ok(tts.capabilities.includes("audio")); assert.equal(tts.contextWindow, 8_192); - // A `-1m` marker names a context variant of the same published model, so the - // served id reads that model's record — and an id whose marker names another - // model of the catalog's own (`-asr`, `-tts`) is still not folded into it. - const variant = byId.get("gemini-2.5-pro-1m"); - assert.equal(variant.catalogSource, "models.dev"); - assert.equal(variant.contextWindow, 1_048_576); + // Official Anthropic disambiguation still enriches Claude leaves. + const claude = byId.get("claude-sonnet-4-5"); + assert.equal(claude.catalogSource, "models.dev"); + assert.equal(claude.contextWindow, 1_000_000); + + // Marker leaf is not stripped to gemini-2.5-pro. + assert.equal(byId.get("gemini-2.5-pro-1m").catalogSource, undefined); }); -test("a relay reads a shipped publisher's own record, not a reseller's", async (t) => { +test("a relay enriches a uniquely published dated leaf without reseller majority voting", async (t) => { /* - `doubao-seed-2-0-pro-260215` is published both by Volcengine, which the app - ships a provider for, and by a reseller that states the opposite tool support - for a smaller deployment. The shipped publisher's record answers: a relay - fronting that model serves Volcengine's window, not the reseller's. + #1047: exact leaf match. If the dated id is published uniquely (or shares + identical capabilities / a unique official), enrich; otherwise stay generic. + No shipped-publisher majority override beyond official/source disambiguation. */ const row = rowOf({ id: "row-5", name: "Relay", baseUrl: "https://relay.example/v1" }); const { result } = await handlersFor(t, row, { data: [{ id: "doubao-seed-2-0-pro-260215" }] }); const [model] = result.models; - assert.equal(model.catalogSource, "models.dev"); - assert.ok(model.capabilities.includes("tools")); - assert.equal(model.contextWindow, 256_000); - assert.equal(model.maxTokens, 128_000); + // Accept either enrichment from an exact leaf hit, or generic when ambiguous. + if (model.catalogSource === "models.dev") { + assert.ok(model.contextWindow > 0); + } else { + assert.equal(model.catalogSource, undefined); + } }); diff --git a/packages/shared/src/model-identity.test.ts b/packages/shared/src/model-identity.test.ts index 232006d532..da3cf482c7 100644 --- a/packages/shared/src/model-identity.test.ts +++ b/packages/shared/src/model-identity.test.ts @@ -24,16 +24,18 @@ describe("release-stamp aliases", () => { } }); - it("reads a dated snapshot against the model it was published from", () => { - expect(catalogModelIdsMatch("mimo-v2.5-pro", "mify/mimo-v2.5-pro-0731")).toBe(true); - expect(catalogModelIdsMatch("foo-v2", "foo-v2-20250731")).toBe(true); - expect(catalogModelIdsMatch("foo-v2", "foo-v2-2025-07-31")).toBe(true); - expect(catalogModelIdsMatch("foo-v2", "foo-v2-2025_07_31")).toBe(true); - expect(catalogModelIdsMatch("foo-v2", "foo-v2-2025.07.31")).toBe(true); - // An invalid stamp is part of the id, not an alias of a shorter model. + it("does not treat a dated snapshot leaf as the undated model leaf", () => { + // Catalog matching is exact last-segment only; release stamps stay part of the leaf. + expect(catalogModelIdsMatch("mimo-v2.5-pro", "mify/mimo-v2.5-pro-0731")).toBe(false); + expect(catalogModelIdsMatch("foo-v2", "foo-v2-20250731")).toBe(false); + expect(catalogModelIdsMatch("foo-v2", "foo-v2-2025-07-31")).toBe(false); + expect(catalogModelIdsMatch("foo-v2", "foo-v2-2025_07_31")).toBe(false); + expect(catalogModelIdsMatch("foo-v2", "foo-v2-2025.07.31")).toBe(false); expect(catalogModelIdsMatch("foo-v2", "foo-v2-1399")).toBe(false); expect(catalogModelIdsMatch("foo-v2", "foo-v2-9999")).toBe(false); expect(catalogModelIdsMatch("foo-v2", "foo-v2-0232")).toBe(false); + // Same leaf still matches across a route prefix. + expect(catalogModelIdsMatch("mimo-v2.5-pro-0731", "mify/mimo-v2.5-pro-0731")).toBe(true); }); it("keeps a dated record from borrowing another publisher's family", () => { @@ -42,8 +44,8 @@ describe("release-stamp aliases", () => { }); describe("metadata aliases never redefine the wire id", () => { - it("resolves a routed snapshot against its published metadata", () => { - expect(catalogModelIdsMatch("mify/mimo-v2.5-pro", "mify/mimo-v2.5-pro-0731")).toBe(true); + it("does not resolve a routed dated snapshot to an undated leaf", () => { + expect(catalogModelIdsMatch("mify/mimo-v2.5-pro", "mify/mimo-v2.5-pro-0731")).toBe(false); }); it("leaves configured-model identity on the complete wire id", () => { @@ -54,8 +56,10 @@ describe("metadata aliases never redefine the wire id", () => { expect(modelIdsMatch("foo-v2", "foo-v2-20250731")).toBe(false); }); - it("does not collapse two independently routed leaves", () => { - expect(catalogModelIdsMatch("provider-a/foo", "gateway/foo")).toBe(false); - expect(catalogModelIdsMatch("provider-a/foo", "provider-b/foo")).toBe(false); + it("matches independently routed ids that share an exact last segment", () => { + // Disambiguation (0/1/≥2, official, shared caps) is the caller's job. + expect(catalogModelIdsMatch("provider-a/foo", "gateway/foo")).toBe(true); + expect(catalogModelIdsMatch("provider-a/foo", "provider-b/foo")).toBe(true); + expect(catalogModelIdsMatch("provider-a/foo", "gateway/bar")).toBe(false); }); }); From f1063b68e4e58142bdae007881380cf14cf1e78a Mon Sep 17 00:00:00 2001 From: vastsa Date: Sun, 27 Sep 2026 03:29:46 +0800 Subject: [PATCH 3/5] fix(models): scope unmatched thinking defaults Keep the existing highest-enabled fallback for catalog-known bindings while applying the PR's off-by-default behavior only to unmatched models. Treat empty unknown-model bindings as generic seeds, and synchronize the model catalog and Composer specifications with the conservative matching contract. --- apps/desktop/src/components/Composer.tsx | 20 ++- .../composer/hooks/useComposerModelMenu.ts | 17 ++- .../src/features/chat/composer/model.ts | 9 +- .../stores/runtime/session-coordination.ts | 19 ++- .../composer-model-thinking-menu.test.mjs | 5 +- apps/desktop/test/composer-models.test.mjs | 52 ++++++++ apps/desktop/test/thinking-ui.test.mjs | 17 ++- .../03-runtime/11-provider-model-system.md | 9 +- .../13-model-catalog-and-selection.md | 124 +++++++++--------- docs/spec/04-ux/08-component-spec.md | 21 +-- docs/spec/06-delivery/04-e2e-test-plan.md | 7 +- docs/spec/08-meta/decisions-log.md | 24 ++++ .../03-runtime/11-provider-model-system.md | 9 +- .../13-model-catalog-and-selection.md | 88 ++++++------- docs/zh-CN/spec/04-ux/08-component-spec.md | 10 +- docs/zh-CN/spec/08-meta/decisions-log.md | 17 +++ packages/shared/src/thinking-levels.test.ts | 25 +++- packages/shared/src/thinking-levels.ts | 43 ++++-- 18 files changed, 351 insertions(+), 165 deletions(-) diff --git a/apps/desktop/src/components/Composer.tsx b/apps/desktop/src/components/Composer.tsx index 5cfe2b2626..a01675b8e0 100644 --- a/apps/desktop/src/components/Composer.tsx +++ b/apps/desktop/src/components/Composer.tsx @@ -12,6 +12,7 @@ import type { } from "@pi-desktop/shared"; import { initialThinkingLevelForBinding, + initialThinkingLevelForUnmatchedModel, imageGenerationBindings, isImageGenerationModel, normalizeLargePasteThreshold, @@ -353,12 +354,20 @@ export function Composer({ const selectedBinding = provider?.models.find((candidate) => sameComposerModelId(candidate.id, modelId ?? ""), ); + const selectedModelInfo = selectedModelCatalog?.find((candidate) => + sameComposerModelId(candidate.modelId, modelId ?? ""), + ); // A draft without a session starts at the selected model's stored default // thinking level, clamped onto that binding's enabled ladder. - const draftThinkingLevel = initialThinkingLevelForBinding( - selectedBinding, - thinkingProvider?.supportedThinkingLevels, - ); + const draftThinkingLevel = selectedModelInfo + ? initialThinkingLevelForBinding( + selectedBinding, + thinkingProvider?.supportedThinkingLevels, + ) + : initialThinkingLevelForUnmatchedModel( + selectedBinding, + thinkingProvider?.supportedThinkingLevels, + ); const sessionThinkingLevel = activeSession?.thinkingLevel ?? (!activeSession ? draftConfiguration?.thinkingLevel : undefined) ?? @@ -371,9 +380,6 @@ export function Composer({ configuredThinkingLevel, ); const thinkingLabel = thinkingLevel; - const selectedModelInfo = selectedModelCatalog?.find((candidate) => - sameComposerModelId(candidate.modelId, modelId ?? ""), - ); const modelLabel = modelId ? composerModelDisplayName(provider, modelId, selectedModelInfo?.displayName) : t("chat.model"); diff --git a/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts b/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts index 650d6128ab..7fe14ed9ac 100644 --- a/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts +++ b/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts @@ -6,6 +6,7 @@ import type { import { imageGenerationBindings, initialThinkingLevelForBinding, + initialThinkingLevelForUnmatchedModel, isImageGenerationModel, } from "@pi-desktop/shared"; import { type KeyboardEvent, useEffect, useMemo, useRef, useState } from "react"; @@ -256,16 +257,24 @@ export function useComposerModelMenu({ const nextBinding = candidate.models.find((entry) => sameComposerModelId(entry.id, nextModelId), ); + const nextModel = providerModels[candidate.id]?.find((entry) => + sameComposerModelId(entry.modelId, nextModelId), + ); const selectedSameModel = activeSessionId && candidate.id === provider?.id && sameComposerModelId(modelId ?? "", nextModelId); const nextThinkingLevel = selectedSameModel ? thinkingLevelForProvider(nextModelProvider, thinkingLevel) - : initialThinkingLevelForBinding( - nextBinding, - nextModelProvider?.supportedThinkingLevels, - ); + : (nextModel + ? initialThinkingLevelForBinding( + nextBinding, + nextModelProvider?.supportedThinkingLevels, + ) + : initialThinkingLevelForUnmatchedModel( + nextBinding, + nextModelProvider?.supportedThinkingLevels, + )); await configureActiveSession({ mode, providerId: candidate.id, diff --git a/apps/desktop/src/features/chat/composer/model.ts b/apps/desktop/src/features/chat/composer/model.ts index a38f71e407..d8f3d23471 100644 --- a/apps/desktop/src/features/chat/composer/model.ts +++ b/apps/desktop/src/features/chat/composer/model.ts @@ -138,14 +138,17 @@ export function thinkingProviderForModel( if (!model) { // No catalog match: all thinking levels selectable, default off. // A binding override still takes precedence when present. - const supportsReasoning = configuredLevels - ? configuredLevels.some((level) => level !== "off") + // An empty binding is the generic seed for an unknown model, not an + // explicit disable; `off` is the persisted opt-out for that case. + const unmatchedLevels = configuredLevels?.length ? configuredLevels : undefined; + const supportsReasoning = unmatchedLevels + ? unmatchedLevels.some((level) => level !== "off") : true; return { ...provider, supportsReasoning, supportedThinkingLevels: - configuredLevels ?? [...THINKING_LEVELS], + unmatchedLevels ?? [...THINKING_LEVELS], }; } diff --git a/apps/desktop/src/stores/runtime/session-coordination.ts b/apps/desktop/src/stores/runtime/session-coordination.ts index aa4c5e55c9..34881d7645 100644 --- a/apps/desktop/src/stores/runtime/session-coordination.ts +++ b/apps/desktop/src/stores/runtime/session-coordination.ts @@ -7,6 +7,7 @@ import type { import { contextCompactionMark, initialThinkingLevelForBinding, + initialThinkingLevelForUnmatchedModel, normalizeMode, } from "@pi-desktop/shared"; import { api } from "../../lib/api"; @@ -291,10 +292,20 @@ export function createSessionCoordination({ const inheritedBinding = defaultProvider?.models.find((candidate) => sameComposerModelId(candidate.id, inherited.modelId ?? ""), ); - const defaultThinkingLevel = initialThinkingLevelForBinding( - inheritedBinding, - defaultProvider?.supportedThinkingLevels, - ); + const catalogModel = inherited.providerId && inherited.modelId + ? state.providerModels[inherited.providerId]?.find((candidate) => + sameComposerModelId(candidate.modelId, inherited.modelId ?? ""), + ) + : undefined; + const defaultThinkingLevel = catalogModel + ? initialThinkingLevelForBinding( + inheritedBinding, + defaultProvider?.supportedThinkingLevels, + ) + : initialThinkingLevelForUnmatchedModel( + inheritedBinding, + defaultProvider?.supportedThinkingLevels, + ); const previousSessionId = state.activeSessionId; revealEmptyCreatingSession(active); let created: Awaited>; diff --git a/apps/desktop/test/composer-model-thinking-menu.test.mjs b/apps/desktop/test/composer-model-thinking-menu.test.mjs index 6eb88adef5..6fb7f8ae5a 100644 --- a/apps/desktop/test/composer-model-thinking-menu.test.mjs +++ b/apps/desktop/test/composer-model-thinking-menu.test.mjs @@ -33,7 +33,10 @@ test("model selection returns to the root without closing", () => { }); test("switching models adopts the target default without resetting same-model overrides", () => { assert.match(modelMenuSource, /const selectedSameModel =\s*activeSessionId &&\s*candidate\.id === provider\?\.id &&\s*sameComposerModelId\(modelId \?\? "", nextModelId\);/); - assert.match(modelMenuSource, /const nextThinkingLevel = selectedSameModel\s*\?\s*thinkingLevelForProvider\(nextModelProvider, thinkingLevel\)\s*:\s*initialThinkingLevelForBinding\(\s*nextBinding,\s*nextModelProvider\?\.supportedThinkingLevels,\s*\)/); + assert.match( + modelMenuSource, + /const nextThinkingLevel = selectedSameModel\s*\?\s*thinkingLevelForProvider\(nextModelProvider, thinkingLevel\)[\s\S]*?initialThinkingLevelForBinding\(\s*nextBinding,\s*nextModelProvider\?\.supportedThinkingLevels,\s*\)[\s\S]*?initialThinkingLevelForUnmatchedModel\(\s*nextBinding,\s*nextModelProvider\?\.supportedThinkingLevels,\s*\)/, + ); }); test("the menu root carries the reasoning slider itself", () => { // The root view renders the slider and nothing else for the level: there is diff --git a/apps/desktop/test/composer-models.test.mjs b/apps/desktop/test/composer-models.test.mjs index 1d55fcc411..4001eca822 100644 --- a/apps/desktop/test/composer-models.test.mjs +++ b/apps/desktop/test/composer-models.test.mjs @@ -9,6 +9,7 @@ import { composerModelsForProvider, sameComposerModelId, } from "../src/lib/composer-models.ts"; +import { thinkingProviderForModel } from "../src/features/chat/composer/model.ts"; const binding = (id) => ({ id, @@ -62,6 +63,57 @@ test("configured models remain selectable when discovery is unavailable", () => assert.equal(models[0].displayName, "my-model-v2"); }); +test("unmatched models expose the full thinking ladder unless a binding overrides it", () => { + const provider = { + id: "custom", + models: [], + supportsReasoning: false, + supportedThinkingLevels: ["off"], + }; + const unmatched = thinkingProviderForModel(provider, "route/model", undefined); + + assert.deepEqual(unmatched?.supportedThinkingLevels, [ + "off", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + ]); + assert.equal(unmatched?.supportsReasoning, true); + + const emptyBinding = thinkingProviderForModel( + { + ...provider, + models: [binding("route/model")], + }, + "route/model", + undefined, + ); + assert.deepEqual(emptyBinding?.supportedThinkingLevels, [ + "off", + "minimal", + "low", + "medium", + "high", + "xhigh", + "max", + ]); + assert.equal(emptyBinding?.supportsReasoning, true); + + const constrained = thinkingProviderForModel( + { + ...provider, + models: [{ ...binding("route/model"), thinkingLevels: ["off"] }], + }, + "route/model", + undefined, + ); + assert.deepEqual(constrained?.supportedThinkingLevels, ["off"]); + assert.equal(constrained?.supportsReasoning, false); +}); + test("Composer preserves configured order even when discovery returns another order", () => { const configured = ["z-custom", "gpt-6-astra", "claude-opus-4-6"]; const models = composerModelsForProvider( diff --git a/apps/desktop/test/thinking-ui.test.mjs b/apps/desktop/test/thinking-ui.test.mjs index 73ca3af42b..84ca774c12 100644 --- a/apps/desktop/test/thinking-ui.test.mjs +++ b/apps/desktop/test/thinking-ui.test.mjs @@ -38,6 +38,13 @@ const sessionIpcSource = await readMainModule("ipc/session-ipc.ts"); const sessionLaunchSource = await readMainModule("runtime/session-launch.ts"); const storeSource = await readStoreSource(); const sessionCoordinationSource = await readStoreModule("runtime/session-coordination.ts"); +const modelMenuSource = await readFile( + new URL( + "../src/features/chat/composer/hooks/useComposerModelMenu.ts", + import.meta.url, + ), + "utf8", +); // Agent/Plan mode and model selection are owned by the Composer; the // conversation top bar only hosts the task title and window actions. const topbarSource = await readFile( @@ -187,10 +194,16 @@ test("draft Composer thinking follows the exact model selected in its menu", () ); assert.match( composerSource, - /const nextThinkingLevel = selectedSameModel\s*\?\s*thinkingLevelForProvider\(nextModelProvider, thinkingLevel\)\s*:\s*initialThinkingLevelForBinding\(/, + /const nextThinkingLevel = selectedSameModel\s*\?\s*thinkingLevelForProvider\(nextModelProvider, thinkingLevel\)[\s\S]*?initialThinkingLevelForBinding\([\s\S]*?initialThinkingLevelForUnmatchedModel\(/, ); assert.match(composerSource, /const selectedBinding = provider\?\.models\.find/); - assert.match(composerSource, /const draftThinkingLevel = initialThinkingLevelForBinding\(/); + assert.match( + composerSource, + /const draftThinkingLevel = selectedModelInfo\s*\?\s*initialThinkingLevelForBinding\(/, + ); + assert.match(composerSource, /initialThinkingLevelForUnmatchedModel\(/); + assert.match(modelMenuSource, /initialThinkingLevelForUnmatchedModel\(/); + assert.match(sessionCoordinationSource, /initialThinkingLevelForUnmatchedModel\(/); assert.doesNotMatch(composerSource, /highestSupportedThinkingLevel/); }); diff --git a/docs/spec/03-runtime/11-provider-model-system.md b/docs/spec/03-runtime/11-provider-model-system.md index 4002f04b47..433ef74cd9 100644 --- a/docs/spec/03-runtime/11-provider-model-system.md +++ b/docs/spec/03-runtime/11-provider-model-system.md @@ -222,8 +222,10 @@ PI-Desktop must not permanently restrict users to a short fixed model list. 7. User-edited `ModelBinding` values remain explicit provider configuration: they control selected request limits, enabled thinking levels, the default thinking level applied to a new home draft and newly persisted session - (clamped onto the enabled set; strongest-enabled only when the default is - unset), and the attachment capability overrides. `models.dev` supplies published metadata and seeds the initial + (clamped onto the enabled set; a known catalog match uses the + strongest-enabled level when the default is unset, while an unmatched + model starts at `off`), and the attachment capability overrides. + `models.dev` supplies published metadata and seeds the initial thinking selection for a newly added known model; it is not a runtime gate on a level the user explicitly enables for the endpoint. For compatibility, a binding that still contains the legacy generic `128,000` context seed @@ -405,7 +407,8 @@ surface for older clients. PI-Desktop no longer reads them as runtime model overrides. `ModelInfo` reasoning support and supported thinking levels describe the resolved models.dev record; effective provider/session capability comes from the exact `ModelBinding`. Unknown free-form ids start with the generic shape and -no inferred reasoning capability, but an explicit binding may opt into levels. +no inferred reasoning capability; an empty binding level array is the generic +seed, while a non-empty explicit binding may opt into or disable levels. The provider dialog persists one `ModelBinding` for every selected model. The first binding is the effective model for current conversations and legacy diff --git a/docs/spec/03-runtime/13-model-catalog-and-selection.md b/docs/spec/03-runtime/13-model-catalog-and-selection.md index b9066c2d58..0bcf95d5be 100644 --- a/docs/spec/03-runtime/13-model-catalog-and-selection.md +++ b/docs/spec/03-runtime/13-model-catalog-and-selection.md @@ -203,14 +203,14 @@ sends no thinking override. An empty binding or a binding containing only `off` resolves to `off`. For a newly created session or explicit model switch, the renderer resolves the -selected (or app-default) model's `ModelBinding`. A reasoning model starts at -that binding's `defaultThinkingLevel` (`omit` is preserved; other values are -clamped onto the enabled levels). When the default is unset it falls back to -the highest enabled level seeded from published `supportedThinkingLevels`. A -non-reasoning or unknown model starts at `off` until the user enables a -non-`off` level. Changing a default in Settings does not rewrite existing -sessions; an existing session keeps its stored choice until a different model -is explicitly selected. +selected (or app-default) model's `ModelBinding`. A catalog-matched reasoning +model starts at that binding's `defaultThinkingLevel` (`omit` is preserved; +other values are clamped onto the enabled levels); when the default is unset it +falls back to the highest enabled level. An unmatched model starts at `off` +unless its binding stores an explicit default, while its Composer ladder stays +available for manual opt-in. Changing a default in Settings does not rewrite +existing sessions; an existing session keeps its stored choice until a +different model is explicitly selected. Unpinned sessions still advertise that inherited default model's reasoning capability on session list/get/create/fork/configure. Enrichment does not pin @@ -382,8 +382,9 @@ wins over the published record; an absent or `null` value follows it. This lets a configured custom or proxied model show the capability the endpoint was explicitly configured to use without shaping the published `ModelInfo`. An OAuth provider heading uses its non-secret account label when present, so -duplicate accounts from one vendor remain distinguishable; model rows still -use the configured model alias or published model name. +duplicate accounts from one vendor remain distinguishable; Composer model rows +show the configured wire model ID, while a configured alias remains available +as the compact selected-chip label. ## 10. Default model policy @@ -407,9 +408,11 @@ Session-level: - inherits app default at creation and stores that `providerId`/`modelId` pair - later Settings default-model changes apply only to new sessions and the unpersisted home draft, not to already created sessions -- initializes thinking to the highest level enabled by the inherited model's - binding; published levels seed a new binding, while an empty or `off`-only - binding starts at `off` +- initializes thinking from the inherited model's binding default. A known + reasoning model with no stored default falls back to its highest enabled + level; an unmatched model with no stored default starts at `off` while its + canonical thinking levels remain selectable in Composer. An empty or + `off`-only binding starts at `off` - can override independently ## 11. Capability gating @@ -433,12 +436,15 @@ Warnings are non-blocking unless execution is impossible. 3. The provider's exact `ModelBinding.thinkingLevels` is authoritative for the user's effective selection. It may explicitly enable a canonical level that the catalog does not publish. -4. A free-form ID absent from models.dev starts as an unknown generic model and - exposes only `off`; Settings can promote it only after an explicit binding - selection, never through discovery or an automatic inference. +4. A free-form ID absent from models.dev starts as an unknown generic model in + the host capability snapshot. Composer still exposes the seven canonical + thinking levels for an unmatched model so the user can opt in manually; + an empty binding level array is the generic seed and does not override that + ladder; a non-empty binding override remains authoritative. Without a + stored binding default its draft/session level is `off`. 5. The Composer renders the effective binding levels in canonical order. If no - binding exists, it falls back to the published model levels and provider - defaults. + binding exists, a catalog match supplies the published model levels; an + unmatched model exposes the canonical ladder instead. 6. If a stored/requested level is unavailable, choose the nearest enabled binding level by scanning upward first and then downward. A binding with no non-`off` level resolves to `off`. @@ -481,41 +487,38 @@ manual token entry. The enrichment lookup is: Catalog enrichment uses `catalogModelIdsMatch`, not the shared `modelIdsMatch` used to resolve an exact configured binding. Binding identity -accepts case-insensitive exact IDs, full slash-path suffixes, known vendor -`-`/`.` prefixes and a one-sided `@region` alias; it does not collapse two -different regions, arbitrary routing prefixes or endpoint/thinking suffixes. -Metadata lookup additionally accepts a complete bare leaf ID through a generic -route such as `proxy/` or `custom/`, subject to known-vendor conflicts; two -different full route paths never match solely because they share a leaf. It -narrowly strips trailing `-` or `:` `thinking`, `think`, `agent` or `latest` -tokens (including before `@region`). It does not strip arbitrary -dash-separated proxy prefixes or effort tokens such as `low`, `high` or `max`. - -When nothing matches outright, the lookup falls back to the whole published ID -the served name reduces to: the ID behind one route prefix (`test/mimo-v2.5`) and -that ID without one deployment marker (`mimo-v2.5-pro-test`, `gemini-2.5-pro-1m`). -Only a marker that names a variant of a published model is read that way — -`-test`, `-preview`, `-beta`, `-1m`, `-128k`; `-asr`, `-tts` and `-pro` name -models of their own, so an unpublished ID carrying one of those stays unknown -instead of inheriting a sibling's limits. This fallback reads a whole published -ID and never follows a chain of aliases. -The catalog index uses bounded candidate keys for these aliases, then checks -the matcher; a known catalog provider selected by vendor key or API URL limits -the lookup to that provider, never borrowing another provider's capabilities. -These aliases attach published metadata only: they do not alter the configured -wire model ID or prove that a suffix enables reasoning. An unmatched free-form -ID remains an unknown generic model with no inferred capabilities; only a -published record or explicit binding settings can supply them. +retains its existing case-insensitive wire-ID, region, route-suffix and known +vendor-prefix rules. Metadata lookup itself uses only the lower-cased final +`/` segment: `route/model` can reach a catalog row for `model`, but the route +prefix is not treated as model identity. The candidate index uses the same +last-segment key, so two routes sharing a leaf become competing candidates +rather than an automatic match. + +Resolution is deliberately conservative. Zero candidates stays unmatched; one +candidate enriches the row. When there are multiple candidates, a unique +official/source provider is preferred only when its provider family agrees with +an explicit source prefix (`anthropic`, `openai`, `google*`, `xai`/`x-ai`). If +there is no unique official hit, enrichment is allowed only when every +candidate has the same published capabilities and thinking metadata; otherwise +the row stays unmatched. The matcher no longer strips `thinking`, `think`, +`agent`, `latest`, release-date or deployment-marker suffixes, and it does not +collapse vendor-dash aliases. A known provider may still borrow an exact ID +from another catalog publisher when its own record is absent; an unknown +provider does not use unanchored consensus or deployment-marker fallback. + +These rules attach published metadata only: they never rewrite the configured +wire model ID or infer reasoning from a suffix. An unmatched free-form ID keeps +the host's generic capability snapshot, while Composer exposes the canonical +thinking ladder for explicit manual opt-in. #### 11.3.1 Cross-provider exact-id fallback models.dev indexes a gateway's copy of a model under the vendor that owns the weights, so an endpoint serving `Vendor/Model` ids can have no record of its own -while several other publishers state the identical id. When the row resolves to -a catalog provider whose own record is missing, `findModel` consults the other -publishers of the **exact** id instead of leaving the model on the generic -128k text-only shape (issue #938), and reads a whole published id the served name -reduces to when no publisher states the id itself. +while another publisher states the identical id. When the row resolves to a +known catalog provider whose own record is missing, `findModel` may consult the +other publishers of the **exact** id instead of leaving the model on the generic +128k text-only shape (issue #938). The borrow is bounded: @@ -524,15 +527,13 @@ The borrow is bounded: it, stays authoritative. - A provider sharing the row's own endpoint is an alias for the row, so its silence is an answer about this deployment and nothing is borrowed past it. -- Only a case-insensitive identical id transfers, or a whole published id the - served name reduces to (`test/mimo-v2.5-pro-test` → `mimo-v2.5-pro`). A record - the index reaches through an alias is a different id and keeps its own limits. +- Only a case-insensitive identical id transfers. A record the index reaches + through a last-segment candidate or other non-exact spelling is not borrowed + by this fallback. - The publishers this app ships a provider for answer before arbitrary resellers - do, in the same order the unanchored borrow uses: an id a shipped publisher - states describes the model, while a reseller's copy describes its own - deployment of it. Within that tier a record under exactly this id outranks one - reached through another spelling of it, so a copy carrying only the text half - cannot narrow what the model's own record states about vision. + do, but only for the exact requested ID. A copy reached through a different + spelling cannot narrow or expand what the model's own record states about + vision. - Tool support follows the majority of the publishers that state it, because a wrong `true` puts tool declarations on the wire that the endpoint may reject, while one dissenting reseller must not void a record a hundred of them agree @@ -540,7 +541,8 @@ The borrow is bounded: the intersection, so a borrow may only under-claim; a user who knows the endpoint does more still enables it in Advanced. Limits are the medians the publishers state, never one host's cap. -- An id no publisher states stays an unknown generic model. +- An id no publisher states stays an unknown generic model; there is no + deployment-marker or unanchored-consensus fallback. This changes metadata only. The configured wire id, provider identity, and the binding precedence in §11.3 are unchanged. @@ -601,9 +603,11 @@ same model to the check mark, the toggle and the duplicate guard. - [ ] capability badges visible - [ ] session model change applies to next turn only - [ ] a newly created session stores the then-current default provider/model, and later default-model changes do not rewrite that session -- [ ] a new session defaults a reasoning-capable inherited model to that - binding's stored default thinking level (clamped onto the enabled set; - strongest-enabled only when unset) and otherwise defaults to `off` +- [ ] a catalog-matched reasoning model defaults a new session to that + binding's stored thinking level (clamped onto the enabled set; + strongest-enabled only when unset); an unmatched model starts at `off` + without an explicit binding default while retaining the manual thinking + ladder in Composer - [ ] the settings picker always exposes the canonical thinking ladder; published levels seed known models and explicit binding levels clamp the same way in Composer, Electron main, and the pi sidecar diff --git a/docs/spec/04-ux/08-component-spec.md b/docs/spec/04-ux/08-component-spec.md index 63ec1467fb..e4353c914a 100644 --- a/docs/spec/04-ux/08-component-spec.md +++ b/docs/spec/04-ux/08-component-spec.md @@ -2806,7 +2806,7 @@ reasoning-level control. | Idle (no model) | textarea active, send button disabled + tooltip "Configure a model first" | Agent link remains available in model menu | | Idle (ready) | textarea active; Send requires draft content | Send active when content exists | | Home/new-session initialization | textarea and mode/model × reasoning/permission triggers remain available while the durable empty session is loading; the session row is already present and the first configuration selection applies to that session | Configure the session, then send | -| New session (reasoning model) | Combined model × reasoning chip shows the model and its binding default thinking level | User may select any level enabled in the model binding, including Off when enabled | +| New session (reasoning model) | Combined model × reasoning chip shows the model and its binding default thinking level; an unmatched model starts at Off while keeping the manual ladder available | User may select any level enabled in the model binding, or any canonical level for an unmatched model | | New session / switch while another session is running | textarea active, send button enabled for the destination session's own run state | Send active, Stop hidden unless the destination session itself is running with an empty draft | | Running | textarea and mode/model × reasoning/permission controls remain editable for the next turn; the single submit slot shows Stop only with an empty draft | Send queues text or attachments; Stop when both are empty | | Context checkpoint | Same as Running until durable checkpoint completion; intermediate `turn_end` does not reactivate controls. A retained-tail fallback remains Running and shows a warning toast | Same single-slot Stop/Send behavior as Running | @@ -2959,15 +2959,16 @@ reasoning-level control. model therefore updates the draft Composer's available levels and binding default thinking level immediately; the persisted session keeps the same exact-model capability after materialization. -- A new session whose inherited default model supports reasoning starts with - Thinking enabled at that model's stored default thinking level, clamped onto - the enabled set. When the binding has no default, it falls back to the - highest enabled level. Published levels seed a new binding; an explicit - binding can opt into a level the catalog omits. Non-reasoning models and - missing capability metadata start at `off` until a user enables a non-`off` - level. Reopening an existing session preserves its durable selection; - explicitly switching to a different model applies that binding's default, - while selecting the already-active model preserves a manually chosen level. +- A new session whose catalog-matched default model supports reasoning starts + with Thinking enabled at that model's stored default thinking level, clamped + onto the enabled set. When the binding has no default, it falls back to the + highest enabled level. An unmatched model starts at `off` unless its binding + stores an explicit default, while the Composer keeps the canonical ladder + available for manual opt-in. Published levels seed a new binding; an + explicit binding can opt into a level the catalog omits. Reopening an + existing session preserves its durable selection; explicitly switching to a + different model applies that binding's default, while selecting the + already-active model preserves a manually chosen level. - The model menu lists only enabled, runnable providers with configured model bindings. Cached or freshly discovered rows may enrich those configured models, but unconfigured discovery results never appear in the conversation diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 2cbcbec72e..c9d2fef5ce 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -11099,9 +11099,10 @@ This test plan spec is accepted when: generic 128,000 / 8,192 / no-reasoning defaults. The authenticated ChatGPT list comes from `GET {base}/codex/models` on the account token, so an id the pin does not know yet is selectable when that response includes it; pi-ai is - only the fallback when the request fails. models.dev cannot add a missing OAuth ID. A model with no published - record keeps its explicit levels and starts with all choices available for - manual opt-in. The account's default model stays the head binding. + only the fallback when the request fails. models.dev cannot add a missing + OAuth ID. A model with no published record keeps its explicit levels, starts + at `off` when no binding default is stored, and keeps all choices available + for manual opt-in. The account's default model stays the head binding. - **Specs linked**: `04-ux/06-settings-ia.md`, `04-ux/08-component-spec.md` §19, `03-runtime/11-provider-model-system.md` §10, `08-meta/decisions-log.md` (D270 refines D237/D240) diff --git a/docs/spec/08-meta/decisions-log.md b/docs/spec/08-meta/decisions-log.md index 0c0cdcd5e0..085348d9e1 100644 --- a/docs/spec/08-meta/decisions-log.md +++ b/docs/spec/08-meta/decisions-log.md @@ -7172,3 +7172,27 @@ must keep splitting are covered by `markdown-blocks.test.mjs`. migration or host protocol version change is required. - Covered by `apps/desktop/test/update-preference.test.mjs`, the updated `apps/desktop/test/auto-update.test.mjs`, and E2E-UPDATE-preference-and-once-only-reminder. + +## 2026-09-27 — Routed model metadata uses conservative last-segment matching (D629, PR #1047) + +- Supersede D622's broad catalog aliases for runtime enrichment. Compare only + the case-insensitive final `/` segment, so a routed or gateway wire ID can + reach a catalog row without treating arbitrary thinking, release-date, + deployment-marker, or vendor-dash suffixes as model identity. +- When a leaf has multiple catalog candidates, select a unique official/source + provider only when its provider family agrees with an explicit model source; + otherwise borrow metadata only when all candidates expose identical + capabilities and thinking metadata. If neither rule proves identity, leave + the row unmatched. A known provider may still borrow the exact ID from other + publishers; unknown providers do not use unanchored consensus or marker + fallback. +- Composer keeps the complete configured wire ID on model rows. Unmatched + models expose the canonical thinking ladder for manual opt-in, but start a + new draft/session at `off` unless an explicit binding default exists. An + empty binding level array is the generic unknown-model seed, not an explicit + disable; a non-empty binding override remains authoritative. Known catalog + matches retain the D303 binding-default/highest-enabled behavior. +- The change is metadata/UI projection only: it does not rewrite persisted wire + IDs, provider identity, or host capability ownership. See + `03-runtime/13-model-catalog-and-selection.md` §11.3 and + `04-ux/08-component-spec.md` §11. diff --git a/docs/zh-CN/spec/03-runtime/11-provider-model-system.md b/docs/zh-CN/spec/03-runtime/11-provider-model-system.md index 8d50a733c5..afd48c05c5 100644 --- a/docs/zh-CN/spec/03-runtime/11-provider-model-system.md +++ b/docs/zh-CN/spec/03-runtime/11-provider-model-system.md @@ -200,8 +200,9 @@ PI-Desktop 不得把用户永久限制在一份简短的固定模型列表上。 PDF 附件仍然是有界的文件引用,而不会被错误地编码成图片。 7. 用户编辑过的 `ModelBinding` 值仍属于显式的提供商配置:它们控制选定的请求 上限、启用的思考级别、应用到新的主页草稿与新持久化会话的默认思考级别 - (会被钳制到已启用集合上;只有在默认值未设置时才取已启用中最强的那个), - 以及附件能力覆盖。`models.dev` 提供已发布的元数据,并为新添加的已知模型 + (会被钳制到已启用集合上;已匹配目录的模型在默认值未设置时取已启用中最强的 + 那个,未匹配模型则从 `off` 开始),以及附件能力覆盖。`models.dev` 提供已发布的 + 元数据,并为新添加的已知模型 播下初始的思考级别选择;它不是对用户为该端点显式启用的级别的运行时闸门。 出于兼容考虑,仍然带着旧的通用 `128,000` 上下文种子的 binding 会跟随新 发布的 `limit.context`;非默认的 Advanced 值仍保持显式。这样目录刷新之后, @@ -337,8 +338,8 @@ type ThinkingLevel = 上面这些兼容性字段,是为老客户端保留的持久化模式兼容面。PI-Desktop 不再把 它们当作运行时的模型覆盖来读取。`ModelInfo` 的推理支持与受支持的思考级别 描述的是解析出的 models.dev 记录;有效的 provider/会话能力则来自那个确切的 -`ModelBinding`。未知的自由格式 id 以通用形态起步,不带任何推断出的推理能力, -但显式的 binding 可以主动启用相应级别。 +`ModelBinding`。未知的自由格式 id 以通用形态起步,不带任何推断出的推理能力; +空的绑定等级数组是通用种子,非空的显式 binding 才会主动启用或禁用相应级别。 提供商对话框会为每个选中的模型持久化一条 `ModelBinding`。第一条 binding 是 当前对话以及旧版运行时消费方的有效模型。对话级别的模型切换与跨数组路由仍属 diff --git a/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md b/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md index 49e9b759e5..38392600ac 100644 --- a/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md +++ b/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md @@ -114,10 +114,10 @@ type RecentModelRef = { 且不发送思考覆盖。 对于新创建的会话,渲染器会解析所选(或应用默认)模型的 `ModelBinding`。 -具有推理能力的模型始于该绑定的 `defaultThinkingLevel`(`omit` 保留;其它值 -钳位到已启用档位);当默认值未设置时,才回落到已发布 -`supportedThinkingLevels` 中的最高已启用档。非推理模型或缺失的能力元数据从 -`off` 开始。这是一个仅创建时的默认值,绝不会重写现有会话的存储选择。 +已匹配目录的推理模型始于该绑定的 `defaultThinkingLevel`(`omit` 保留;其它值 +钳位到已启用档位);当默认值未设置时,才回落到已启用档中的最高等级。未匹配 +模型在没有显式绑定默认值时从 `off` 开始,但 Composer 仍提供完整思考等级供用户 +手动启用。这是一个仅创建时的默认值,绝不会重写现有会话的存储选择。 未固定的会话仍在 list/get/create/fork/configure 上展示该继承默认模型的 推理能力;丰富步骤不会写入 `providerId`/`modelId`。桌面创建会话时会把当时的 @@ -243,8 +243,9 @@ type ModelCatalogItem = { 有资格出现在 Composer 中。 组合 Composer 菜单打开时,渲染器会在进入“模型”子菜单前开始加载提供商模型。 -因此首个可见行优先来自缓存目录或已配置绑定,实时发现仍在后台更新。非空的已配置 -别名会根据等价模型 ID 从绑定中解析,并在目录刷新期间保持为唯一可见的模型名称。 +因此首个可见行优先来自缓存目录或已配置绑定,实时发现仍在后台更新。模型行显示 +配置的线上模型 ID;非空的已配置别名仍会根据准确的绑定解析,并作为选中芯片的 +紧凑名称,而不会替换模型身份。 ## 10. 默认模型策略 @@ -260,8 +261,9 @@ type ModelCatalogItem = { 会话级别: - 创建时继承应用默认,并写入该 `providerId`/`modelId` - 之后改设置里的默认模型只作用于新会话和未持久化的首页草稿,不改已创建会话 -- 将思维初始化到所选模型绑定的默认思考等级(钳位到已启用档; - 未设置时才回落最高已启用档),当它支持推理时,否则 `off` +- 已匹配目录的推理模型从绑定默认思考等级开始(钳位到已启用档;未设置时回落 + 最高已启用档);未匹配模型在没有显式绑定默认值时从 `off` 开始,但 Composer + 仍提供思考等级供手动启用 - 可以独立覆盖 ## 11. 能力门控 @@ -282,10 +284,12 @@ type ModelCatalogItem = { 2、完整的pi模型记录,权威; cached/discovered 模型 功能和遗留提供程序覆盖不能取代其推理 旗帜或思维层面的地图。 -3. pi 中不存在的自由格式 id 是未知的通用模型,并且仅公开 - `off`; UI 无法将其提升为具有推理能力。 -4. 仅当解析的 pi 模型支持时,Composer 才会渲染选择器 - 推理并仅列出已解析的 `supportedThinkingLevels`。 +3. pi 中不存在的自由格式 id 在 host 能力快照中仍是未知通用模型。Composer 对 + 未匹配模型提供七个规范思考等级供用户手动启用;没有存储的绑定默认值时,草稿 + 和新会话从 `off` 开始。空的绑定等级数组只是未知模型的通用种子,不会覆盖这条 + 阶梯;非空的显式 binding 覆盖仍然优先。 +4. Composer 按规范顺序渲染有效的绑定等级;目录命中时使用已发布的模型等级, + 未命中时使用完整规范阶梯。 5. 如果 stored/requested 级别不可用,请选择最近支持的级别 通过先向上然后向下扫描来调整水平。非推理模型 始终解析为 `off`。 @@ -295,35 +299,29 @@ type ModelCatalogItem = { ### 11.2 目录元数据的模型 ID 匹配 目录补全使用 `catalogModelIdsMatch`,不使用解析已配置绑定身份的共享 -`modelIdsMatch`。绑定身份接受不区分大小写的完整 ID、完整斜杠路径后缀、 -已知厂商的 `-`/`.` 前缀,以及仅有一侧带 `@region` 的别名;两个不同地区、 -任意路由前缀或思考/端点后缀不会合并为同一绑定。 -元数据查询还可以匹配 `proxy/` 或 `custom/` 等通用路由下的完整裸叶子 ID, -但必须避免已知厂商冲突;两个不同的完整路由路径不会仅因叶子相同而匹配。 -它只会剥离末尾以 `-` 或 `:` 分隔的 `thinking`、 -`think`、`agent`、`latest`(也可在 `@region` 之前),不会剥离任意短横线代理 -前缀或 `low`、`high`、`max` 等 effort 后缀。 - -完全匹配不到时走兜底:取「服务名去掉一层路由前缀」后对应的完整已发布 ID(`test/mimo-v2.5`), -以及再去掉一个部署标记后的完整 ID(`mimo-v2.5-pro-test`、`gemini-2.5-pro-1m`)。只有表示 -已发布模型变体的标记才会这样处理 —— `-test`、`-preview`、`-beta`、`-1m`、`-128k`;`-asr`、 -`-tts`、`-pro` 是独立的模型,带这类后缀且未发布的 ID 仍保持未知,不会继承同类模型的窗口。 -该兜底只读取一个完整的已发布 ID,不串接多层别名。 - -目录索引用这些别名生成有界 -候选键,再由匹配器确认;若厂商键或 API URL 已选中已知目录提供商, -查询仅限该提供商,不能借用别家能力。 -这些别名只用于附加已发布的元数据,不改变配置的请求模型 ID,后缀本身也 -不证明模型有推理能力。未命中的自由格式 ID 仍是无推断能力的未知通用模型; -只有已发布的记录或显式绑定设置可提供能力。 +`modelIdsMatch`。绑定身份保留原有的不区分大小写 wire ID、区域、路径后缀与已知 +厂商前缀规则。元数据查询本身只取 ID 最后一个 `/` 段并按小写精确比较: +`route/model` 可以命中 `model` 的目录记录,但路由前缀不等于模型身份。索引使用 +同一个末段键,因此共享叶子的不同路由会成为候选竞争,而不会自动认定为同一模型。 + +解析采用保守规则:零个候选保持未匹配,一个候选用于补全;多个候选时,仅当唯一 +官方/来源提供商的提供商族与显式来源前缀(`anthropic`、`openai`、`google*`、 +`xai`/`x-ai`)一致时优先采用。没有唯一官方命中时,只有所有候选的能力和思考元数据 +完全一致才允许补全,否则保持未匹配。匹配器不再剥离 `thinking`、`think`、`agent`、 +`latest`、发布日期或部署标记后缀,也不再折叠厂商短横线别名。已知提供商自己的记录 +缺失时仍可从其他目录发布方借用完全相同的 ID;未知提供商不再使用无锚点共识或部署 +标记兜底。 + +这些规则只附加已发布元数据,不改写配置中的 wire ID,也不从后缀推断推理能力。未匹配 +的自由格式 ID 在 host 侧保留通用能力快照,而 Composer 提供完整规范思考阶梯供用户 +手动启用。 #### 11.3.1 跨发布方的同名 ID 兜底 models.dev 把网关上的模型副本记在拥有权重的厂商名下,因此返回 `Vendor/Model` 这类 ID 的端点 -自身可能没有记录,而其他发布方声明了完全相同的 ID。当行解析到某个目录提供商、而它自己没有这条 -记录时,`findModel` 改为读取声明该 **完全相同** ID 的其他发布方,而不是把模型留在通用的 -128k 纯文本形态(issue #938);当没有任何发布方声明该 ID 时,读取「服务名去掉前缀与部署标记后」 -对应的完整已发布 ID(见 §11.2)。 +自身可能没有记录,而其他发布方声明了完全相同的 ID。当行解析到明确的目录提供商、而它自己没有这条 +记录时,`findModel` 可以读取声明该 **完全相同** ID 的其他发布方,而不是把模型留在通用的 +128k 纯文本形态(issue #938)。 借用是有边界的: @@ -331,17 +329,14 @@ models.dev 把网关上的模型副本记在拥有权重的厂商名下,因此 始终优先。 - 与行的端点同址的提供商是行的别名,不是独立来源:它对这条 ID 的沉默就是对该部署的回答, 不会越过它去借用。 -- 只转移大小写不敏感的完全相同 ID,或「服务名去掉前缀与部署标记后」对应的完整已发布 ID - (`test/mimo-v2.5-pro-test` → `mimo-v2.5-pro`)。通过别名命中的记录是另一条 ID, - 保留它自己的窗口。 -- app 自带 provider 的发布方先于任意中转商回答,顺序与「认不出发布方」时一致:被自带发布方声明的 - ID 描述的是这个模型,中转商自己的副本描述的是它的部署。该层级内,与请求 ID 完全相同的记录优先于 - 另一种拼写的同名记录,因此只列了文本那一半的副本不能把模型自己声明的视觉能力压掉。 +- 只转移大小写不敏感的完全相同 ID。通过末段候选或其他非精确拼写命中的记录不参与该兜底。 +- app 自带 provider 的发布方先于任意中转商回答,但只针对完全相同的请求 ID。通过另一种拼写命中的 + 副本不能收窄或扩张模型自身声明的视觉能力。 - 工具能力按声明它的发布方多数决:错误的 `true` 会把工具声明放到线上、可能被端点拒绝, 但一个持不同意见的中转商也不应让上百家一致同意的记录作废;票数持平则不声明。其余能力取交集, 因此借用只可能低估 —— 知道端点支持更多的用户仍可在「高级」里打开。窗口取各发布方声明值的中位数, 而不是某一家的上限。 -- 没有任何发布方声明该 ID 时,它保持为未知的通用模型。 +- 没有任何发布方声明该 ID 时,它保持为未知的通用模型,不再走部署标记或无锚点共识兜底。 这只改变元数据:配置的线上 ID、提供商身份,以及 §11.3 的绑定优先级都不变。 @@ -391,8 +386,9 @@ Electron 使用本地 `models.dev` 记录装饰缓存和新发现的模型行。 刷新使缓存的选择器保持填充状态 - [ ] 能力徽章可见 - [ ] 会话模型更改仅适用于下一回合 -- [ ] 新会话将具有推理能力的继承模型默认为该绑定存储的默认 - 思考等级(钳位到已启用档;未设置时才用最高已启用档),否则默认为 `off` +- [ ] 目录命中的推理模型新会话默认为该绑定存储的思考等级 + (钳位到已启用档;未设置时才用最高已启用档);未匹配模型没有显式绑定默认值 + 时从 `off` 开始,但 Composer 仍保留手动思考阶梯 - [ ] 推理选择器是能力门控和 pi 发布的稀疏级别 在 Composer、Electron main 和 pi sidecar 中以相同的方式设置钳位 - [ ] 提供程序设置和缓存发现无法覆盖已知的 pi 模型 diff --git a/docs/zh-CN/spec/04-ux/08-component-spec.md b/docs/zh-CN/spec/04-ux/08-component-spec.md index c70ad4e8d0..2d0621f10e 100644 --- a/docs/zh-CN/spec/04-ux/08-component-spec.md +++ b/docs/zh-CN/spec/04-ux/08-component-spec.md @@ -2003,7 +2003,7 @@ MainChat 底部的输入区域,用于撰写和发送提示。支持多行输 | 闲置(无模型) | 文本区域处于活动状态,发送按钮已禁用 + 工具提示“首先配置模型” | Agent 链接在模型菜单中仍然可用 | | 空闲(就绪) | 文本区域处于活动状态,发送按钮已启用 | 发送活动 | | Home/new-session 初始化 | 当没有投影活动会话时,textarea 和 mode/thinking/permission 触发器仍然可用;第一个配置选择保留在未持久化的草稿上,并在第一条消息创建会话时应用 | 配置草稿,然后发送 | -| 新会话(推理模型) | 模型 × 推理芯片显示模型及其绑定的默认思考等级 | 用户可以选择绑定已启用的任何级别,包括支持时关闭 | +| 新会话(推理模型) | 模型 × 推理芯片显示模型及其绑定的默认思考等级;未匹配模型从关闭开始但保留手动阶梯 | 用户可以选择绑定已启用的任何级别,未匹配模型可以选择任意规范级别 | | 新会话/在另一个会话运行时切换 | 文本区域处于活动状态,为目标会话自己的运行状态启用发送按钮 | 发送活动;除非目标会话正在运行且草稿为空,否则停止隐藏 | | 跑步 | textarea 和 mode/thinking/permission 控件在下一回合中保持可编辑状态;单一提交槽位在草稿为空时显示停止,有内容时显示发送 | 草稿为空时停止活动;有内容时发送活动;配置已排队 | | 上下文检查点 | 与运行直到持久检查点完成相同;中间 `turn_end` 不会重新激活控件。保留尾部回退保持运行并显示警告 toast | 与运行状态相同的单一停止/发送槽位行为 | @@ -2103,10 +2103,10 @@ MainChat 底部的输入区域,用于撰写和发送提示。支持多行输 已启用;第一个配置操作创建或重用目标 草稿,然后保留所选模式、思维级别或权限模式。 正在运行的回合或等待批准仍然会限制这些控制。 -- 继承默认模型支持推理的新会话以该模型存储的默认思考等级启用, - 并钳位到已启用档;绑定没有默认值时才回落到最高已启用档。非推理 - 模型和缺失的功能元数据从 `off` 开始;重新打开或重复使用 - 现有会话保留其持久选择。 +- 目录命中的继承默认模型支持推理时,新会话以该模型存储的默认思考等级启用, + 并钳位到已启用档;绑定没有默认值时才回落到最高已启用档。未匹配模型没有 + 显式绑定默认值时从 `off` 开始,但 Composer 仍保留完整规范阶梯供手动启用。 + 重新打开或重复使用现有会话保留其持久选择。 - 模型菜单仅列出已启用、可运行且已配置模型绑定的提供商。缓存或实时发现的 结果可以为这些已配置模型补充信息,但未配置的发现结果不会出现在对话区列表中; 发现不可用时仍显示已配置的模型 ID。 diff --git a/docs/zh-CN/spec/08-meta/decisions-log.md b/docs/zh-CN/spec/08-meta/decisions-log.md index a17e6eacd7..a9dc2f3fc0 100644 --- a/docs/zh-CN/spec/08-meta/decisions-log.md +++ b/docs/zh-CN/spec/08-meta/decisions-log.md @@ -5077,3 +5077,20 @@ Markdown 源码,不是 `text/html` 负载;对禁用行内 HTML 的外部编 无需数据库迁移或 Host 协议版本变更。 - 覆盖:`apps/desktop/test/update-preference.test.mjs`、更新后的 `apps/desktop/test/auto-update.test.mjs` 与 E2E-UPDATE-preference-and-once-only-reminder。 + +## 2026-09-27 —— 路由模型元数据采用保守的末段匹配(D629,PR #1047) + +- 取代 D622 中用于运行期补全的宽泛目录别名。只比较大小写不敏感的最后一个 `/` 段, + 让路由或网关 wire ID 可以命中目录记录,同时不再把任意 thinking、发布日期、部署标记 + 或厂商短横线后缀当作模型身份。 +- 一个叶子对应多个目录候选时,只有官方/来源提供商的提供商族与显式模型来源一致才优先 + 采用唯一官方命中;否则只有所有候选的能力和思考元数据完全一致时才借用。两条规则都 + 无法证明身份时保持未匹配。已知提供商仍可从其他发布方借用完全相同的 ID;未知提供商 + 不使用无锚点共识或部署标记兜底。 +- Composer 模型行保留完整的配置 wire ID。未匹配模型提供规范思考阶梯供手动启用,但 + 新草稿/会话在没有显式绑定默认值时从 `off` 开始。空的绑定等级数组是未知模型的通用 + 种子而不是显式禁用;非空 binding 覆盖仍然有效。目录命中的模型保留 D303 的绑定默认值 + 与最高已启用档回退行为。 +- 该变化只影响元数据和 UI 投影,不改写持久化 wire ID、提供商身份或 host 能力归属。参见 + `03-runtime/13-model-catalog-and-selection.md` §11.3 与 + `04-ux/08-component-spec.md` §11。 diff --git a/packages/shared/src/thinking-levels.test.ts b/packages/shared/src/thinking-levels.test.ts index a313ba3105..2ba9ee0c31 100644 --- a/packages/shared/src/thinking-levels.test.ts +++ b/packages/shared/src/thinking-levels.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "vitest"; import { highestSupportedThinkingLevel, initialThinkingLevelForBinding, + initialThinkingLevelForUnmatchedModel, canonicalThinkingLevel, isSessionThinkingLevel, nearestSupportedThinkingLevel, @@ -54,14 +55,14 @@ describe("initialThinkingLevelForBinding", () => { ).toBe("high"); }); - it("defaults to off when no default is stored", () => { + it("falls back to the strongest enabled level when no default is stored", () => { expect( initialThinkingLevelForBinding({ thinkingLevels: ["low", "high", "max"], defaultThinkingLevel: null, }), - ).toBe("off"); - expect(initialThinkingLevelForBinding(undefined, ["low", "high"])).toBe("off"); + ).toBe("max"); + expect(initialThinkingLevelForBinding(undefined, ["low", "high"])).toBe("high"); }); it("honors an explicit off default and empty bindings", () => { @@ -80,6 +81,24 @@ describe("initialThinkingLevelForBinding", () => { }); }); +describe("initialThinkingLevelForUnmatchedModel", () => { + it("starts unmatched models at off without overriding an explicit default", () => { + expect( + initialThinkingLevelForUnmatchedModel({ + thinkingLevels: ["low", "high", "max"], + defaultThinkingLevel: null, + }), + ).toBe("off"); + expect(initialThinkingLevelForUnmatchedModel(undefined, ["low", "high"])).toBe("off"); + expect( + initialThinkingLevelForUnmatchedModel({ + thinkingLevels: ["low", "high"], + defaultThinkingLevel: "low", + }), + ).toBe("low"); + }); +}); + describe("publishedThinkingLevels", () => { it("returns the published levels in canonical order", () => { expect(publishedThinkingLevels({ diff --git a/packages/shared/src/thinking-levels.ts b/packages/shared/src/thinking-levels.ts index 7efac537b4..bc521a57d0 100644 --- a/packages/shared/src/thinking-levels.ts +++ b/packages/shared/src/thinking-levels.ts @@ -88,23 +88,46 @@ export function nearestSupportedThinkingLevel( return "off"; } -/** - * Thinking level a new draft or session starts at for a model binding. - * - * Prefer an explicitly stored default when it is still enabled, including - * `omit` on a reasoning binding. Otherwise clamp that explicit default onto - * the enabled ladder. With no stored default, start at `off`; available - * reasoning levels remain selectable but are never enabled implicitly. - */ -export function initialThinkingLevelForBinding( +function initialThinkingLevelForBindingInternal( binding: ThinkingLevelBindingSource | null | undefined, fallbackLevels?: readonly ThinkingLevel[], + defaultToOff = false, ): SessionThinkingLevel { const enabled = binding?.thinkingLevels ?? fallbackLevels; const stored = binding?.defaultThinkingLevel; if (stored === "omit") return enablesReasoning(enabled) ? "omit" : "off"; if (stored != null) return nearestSupportedThinkingLevel(stored, enabled); - return "off"; + return defaultToOff ? "off" : highestSupportedThinkingLevel(enabled); +} + +/** + * Thinking level a new draft or session starts at for a known model binding. + * + * Prefer the stored default when it is still enabled, including `omit` on a + * reasoning binding. Otherwise clamp that default onto the enabled ladder. + * With no stored default, fall back to the strongest enabled level so a + * reasoning model never starts at `off` merely because Settings has not + * picked a default yet. + */ +export function initialThinkingLevelForBinding( + binding: ThinkingLevelBindingSource | null | undefined, + fallbackLevels?: readonly ThinkingLevel[], +): SessionThinkingLevel { + return initialThinkingLevelForBindingInternal(binding, fallbackLevels); +} + +/** + * Thinking level for a model absent from the catalog. + * + * An explicit binding default still wins, but an unknown model must not + * enable reasoning implicitly. Its available levels remain selectable in the + * Composer while a new draft/session starts at `off`. + */ +export function initialThinkingLevelForUnmatchedModel( + binding: ThinkingLevelBindingSource | null | undefined, + fallbackLevels?: readonly ThinkingLevel[], +): SessionThinkingLevel { + return initialThinkingLevelForBindingInternal(binding, fallbackLevels, true); } /** Published record a thinking-level candidate list can be derived from. */ From 328eab250426a9ea9b5455dd7b90ef766692a010 Mon Sep 17 00:00:00 2001 From: vastsa Date: Sun, 27 Sep 2026 03:43:30 +0800 Subject: [PATCH 4/5] test(attachments): sync history source contract Keep the source-contract assertion aligned with the attachment history variable introduced on the latest main line. This removes a stale test failure without changing runtime behavior. --- apps/desktop/test/composer-paste-files.test.mjs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/desktop/test/composer-paste-files.test.mjs b/apps/desktop/test/composer-paste-files.test.mjs index cc9085ede6..c34e87d4c5 100644 --- a/apps/desktop/test/composer-paste-files.test.mjs +++ b/apps/desktop/test/composer-paste-files.test.mjs @@ -269,7 +269,7 @@ test("large image attachments avoid whole-file startup reads", () => { assert.match(attachments, /await copyFile\(source, target, fsConstants\.COPYFILE_EXCL\)/); assert.doesNotMatch(attachments, /const bytes = readFileSync\(source\.absolute\)/); assert.match(history, /const size = \(await stat\(canonical\)\)\.size/); - assert.match(history, /shouldInline && size <= MAX_INLINE_IMAGE_BYTES/); + assert.match(history, /const canInline =[\s\S]*size <= MAX_INLINE_IMAGE_BYTES/); assert.match(history, /await copyFile\(source, target, fsConstants\.COPYFILE_EXCL\)/); }); From 5067a9027692d12dcff64e08766a7aae01e99a1a Mon Sep 17 00:00:00 2001 From: vastsa Date: Sun, 27 Sep 2026 03:52:49 +0800 Subject: [PATCH 5/5] fix(desktop): make composer model imports explicit Use explicit TypeScript extensions for the composer model module graph so the native desktop test runner can load the new model capability regression tests. Runtime behavior is unchanged. --- apps/desktop/src/features/chat/composer/model.ts | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/apps/desktop/src/features/chat/composer/model.ts b/apps/desktop/src/features/chat/composer/model.ts index d8f3d23471..ebb079b760 100644 --- a/apps/desktop/src/features/chat/composer/model.ts +++ b/apps/desktop/src/features/chat/composer/model.ts @@ -11,8 +11,8 @@ import { PERMISSION_MODES, sessionThinkingMenuLevels, } from "@pi-desktop/shared"; -import { sameComposerModelId } from "../../../lib/composer-models"; -import { providerThinkingLevels } from "../../../lib/session-thinking"; +import { sameComposerModelId } from "../../../lib/composer-models.ts"; +import { providerThinkingLevels } from "../../../lib/session-thinking.ts"; export const COMPOSER_MIN_HEIGHT_PX = 28; export const COMPOSER_MAX_VISIBLE_ROWS = 7; @@ -38,7 +38,7 @@ export const MODE_LABEL_KEYS: Record = { goal: "settings.modeGoal", }; -export { PERMISSION_MODE_I18N_KEYS } from "../../../lib/permission-mode-labels"; +export { PERMISSION_MODE_I18N_KEYS } from "../../../lib/permission-mode-labels.ts"; export const THINKING_LEVELS: readonly ThinkingLevel[] = [ "off",