Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 24 additions & 0 deletions src/provider/chatPrep.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ import {
modelLimits,
resolveRawModelId,
} from "./settings";
import { createHash } from "node:crypto";
import { messagesHaveImages } from "../request/builders";
import { buildOpenCodeRequestHeaders } from "../request/headers";
import type { ApiMessage, ApiSettings, OpenAiContentPart } from "../request/types";
Expand Down Expand Up @@ -281,6 +282,29 @@ export async function prepareChatRequest(
// call), so creating one per request flooded the Output tab with dozens of
// duplicate "OpenCode" channels (issue #220). Transports receive the
// provider's shared channel instead.
{
const flag = process.env.OPENCODE_CONTEXT_CACHE_DEBUG ?? "";
const debugEnabled = flag === "1" || flag === "true";
if (debugEnabled) {
try {
const prefixAnchor = apiMessages
.slice(0, 3)
.map((m) => `${m.role}:${JSON.stringify(m.content).slice(0, 2048)}`)
.join("\n");
const toolNames = [...(options.tools ?? [])]
.map((t) => (t as { name: string }).name)
.sort((a, b) => (a < b ? -1 : a > b ? 1 : 0))
.join(",");
const prefixHash = createHash("sha256")
.update(prefixAnchor + "|" + toolNames, "utf8")
.digest("hex")
.slice(0, 12);
deps.log(`[prefix-hash] ${prefixHash} messages=${String(apiMessages.length)} tools=${String(options.tools?.length ?? 0)}`);
} catch (e) {
deps.log(`[prefix-hash] failed: ${String(e)}`);
}
}
}
const onTransportSummary = (summary: TransportRequestSummary) => {
// Compute credits for VS Code session cost (1 credit = $0.01).
// VS Code reads usage.copilotCredits from the LanguageModelDataPart
Expand Down
3 changes: 2 additions & 1 deletion src/request/anthropic.ts
Original file line number Diff line number Diff line change
Expand Up @@ -221,7 +221,8 @@ function anthropicImageSource(part: OpenAiContentPart): AnthropicImageSource | u
}

function mapAnthropicTools(tools: readonly vscode.LanguageModelChatTool[] | undefined): AnthropicToolDefinition[] {
return (tools ?? []).map((tool) => ({
const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
return sorted.map((tool) => ({
name: tool.name,
description: tool.description,
input_schema: sanitizeToolSchema(tool.inputSchema),
Expand Down
3 changes: 2 additions & 1 deletion src/request/google.ts
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,8 @@ export function buildGoogleGenerateContentBody(
}

function mapGoogleTools(tools: readonly vscode.LanguageModelChatTool[] | undefined): Record<string, unknown>[] {
return (tools ?? []).map((tool) => ({
const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
return sorted.map((tool) => ({
name: tool.name,
description: tool.description,
parameters: sanitizeToolSchema(tool.inputSchema),
Expand Down
75 changes: 73 additions & 2 deletions src/request/headers.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,8 @@
import * as vscode from "vscode";
import { randomUUID } from "node:crypto";
import { createHash, randomUUID } from "node:crypto";
import * as os from "os";
import * as path from "path";
import * as fs from "fs";
import { OPEN_CODE_CLIENT } from "../config";
import { getUserAgent } from "../provider/definitions";
import { messageText } from "../provider/tokens";
Expand Down Expand Up @@ -30,6 +33,67 @@ export function auxiliarySessionId(context: vscode.ExtensionContext): string {
return id;
}

// --- Context-cache parity (mirrors ~/.config/opencode/plugins/opencode-context-cache.mjs) ---
// NOTE (PR #212 review): the legacy model headers x-session-id /
// conversation_id / session_id were dropped — no evidence they do anything
// beyond x-opencode-session + prompt_cache_key, and they are not in any
// public Zen docs. The project cache key now flows only via prompt_cache_key
// (chat-completions/responses bodies); x-opencode-session stays the routing
// affinity header.
const CONTEXT_CACHE_DEBUG_ENV_VAR = "OPENCODE_CONTEXT_CACHE_DEBUG";

function appendContextCacheLog(message: string): void {
const flag = process.env[CONTEXT_CACHE_DEBUG_ENV_VAR] ?? "";
if (flag !== "1" && flag !== "true") return;
try {
const logPath = path.join(os.homedir(), ".config", "opencode", "plugins", "context-cache-vscode.log");
const safe = message.replace(/\n/g, "\\n").replace(/\r/g, "\\r");
const line = `[${new Date().toISOString()}] [pid:${String(process.pid)}] [context-cache-vscode] ${safe}\n`;
fs.appendFileSync(logPath, line, "utf8");
} catch {
/* best-effort */
}
}

export function hashRawCacheKey(raw: string): string {
return createHash("sha256").update(raw, "utf8").digest("hex");
}

function normalizeDirForCacheKey(dir: string): string {
// Canonicalize separators so C:\a\b and C:/a/b hash identically.
// Drive-letter upper-casing keeps c:\ vs C:\ stable on Windows.
let out = dir.replace(/\\/g, "/");
if (out.length >= 2 && out[1] === ":" && out[0] !== out[0].toUpperCase()) out = out[0].toUpperCase() + out.slice(1);
return out;
}

function resolveRawProjectCacheKey(modelId: string): string | null {
const env = process.env as Record<string, string | undefined>;
const override = (env.OPENCODE_PROMPT_CACHE_KEY ?? env.OPENCODE_STICKY_SESSION_ID ?? "").trim();
if (override) return override;
try {
const user = env.USERNAME ?? env.USER ?? env.LOGNAME ?? "unknown";
const host = os.hostname();
let dir = "";
try {
dir = vscode.workspace.workspaceFolders?.[0]?.uri.fsPath ?? "";
} catch {
dir = "";
}
if (dir) dir = normalizeDirForCacheKey(dir);
if (!dir) dir = modelId || "no-workspace";
return `${user}@${host}:${dir}`;
} catch {
return null;
}
}

export function resolveProjectCacheKey(modelId: string): string | null {
const raw = resolveRawProjectCacheKey(modelId);
if (!raw) return null;
return hashRawCacheKey(raw);
}

// The official OpenCode client sends these headers on every request. The Zen
// gateway reads x-opencode-session first, then converts that sticky identifier
// into provider-specific affinity headers such as x-session-affinity upstream.
Expand Down Expand Up @@ -62,12 +126,19 @@ export function buildOpenCodeRequestHeaders(
`req-${stableHash(`${String(Date.now())}-${String(Math.random())}-${sessionId}-${modelId}`)}`,
);

return {
const projectCacheKey = resolveProjectCacheKey(modelId);
const headers: Record<string, string> = {
"x-opencode-session": sessionId,
"x-opencode-request": requestId,
"x-opencode-client": OPEN_CODE_CLIENT,
"User-Agent": getUserAgent(),
};
if (projectCacheKey) {
appendContextCacheLog(`model=${modelId} raw=${resolveRawProjectCacheKey(modelId) ?? ""} hash=${projectCacheKey}`);
} else {
appendContextCacheLog(`model=${modelId} no stable cache key resolved`);
}
return headers;
}

/**
Expand Down
15 changes: 13 additions & 2 deletions src/request/openai.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ import { lookupModelRegistryEntry } from "../core/registry";
import { buildResponsesRequestEnvelope, pairResponsesFunctionCallItems, responsesInputItemsFromMessage } from "../responsesRequest";
import { thinkingProviderFor } from "../thinking";
import { sanitizeToolSchema } from "./schema";
import { resolveProjectCacheKey } from "./headers";
import { messagesHaveImages } from "./shared";
import type { ResolvedModelMetadata } from "../models/metadata";
import type { ModelLimits } from "../models/modelLimits";
Expand Down Expand Up @@ -39,6 +40,10 @@ export function buildChatCompletionsRequestBody(
max_tokens: limits.maxOutputTokens,
stream: true,
stream_options: { include_usage: true },
...(() => {
const k = resolveProjectCacheKey(modelId);
return k ? { prompt_cache_key: k } : {};
})(),
...thinkingPayload,
...(tools.length ? { tools, tool_choice: toolChoice(options.toolMode) } : {}),
};
Expand Down Expand Up @@ -66,6 +71,10 @@ export function buildResponsesRequestBody(
model: modelId,
input,
maxOutputTokens: limits.maxOutputTokens,
...(() => {
const k = resolveProjectCacheKey(modelId);
return k ? { promptCacheKey: k } : {};
})(),
// Muse Spark gateway requires `truncation: "disabled"` — requests whose
// input exceeds the 1M context window will hard-fail (HTTP 400) instead
// of being silently truncated upstream. This is a gateway constraint,
Expand All @@ -80,7 +89,8 @@ export function buildResponsesRequestBody(
}

function mapOpenAiTools(tools: readonly vscode.LanguageModelChatTool[] | undefined): OpenAiToolDefinition[] {
return (tools ?? []).map((tool) => ({
const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
return sorted.map((tool) => ({
type: "function",
function: {
name: tool.name,
Expand All @@ -92,7 +102,8 @@ function mapOpenAiTools(tools: readonly vscode.LanguageModelChatTool[] | undefin

function mapResponsesTools(tools: readonly vscode.LanguageModelChatTool[] | undefined, modelId?: string): Record<string, unknown>[] {
const needsTruncation = isMuseFamily(modelId ?? "");
return (tools ?? []).map((tool) => ({
const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
return sorted.map((tool) => ({
type: "function",
name: needsTruncation ? truncateToolName(tool.name) : tool.name,
description: tool.description,
Expand Down
2 changes: 1 addition & 1 deletion src/request/schema.ts
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,7 @@ function sanitizeJsonSchemaNode(value: unknown, root: Record<string, unknown>, s
}

const result: Record<string, unknown> = {};
for (const [key, child] of Object.entries(value)) {
for (const [key, child] of Object.entries(value).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) {
if (key === "$schema" || key === "$id" || key === "$ref" || key === "$defs" || key === "definitions") {
continue;
}
Expand Down
2 changes: 2 additions & 0 deletions src/responsesRequest.ts
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@ export interface ResponsesRequestEnvelopeOptions {
tools?: readonly unknown[];
toolChoice?: unknown;
truncation?: "auto" | "disabled";
promptCacheKey?: string;
}

/**
Expand Down Expand Up @@ -46,6 +47,7 @@ export function buildResponsesRequestEnvelope(options: ResponsesRequestEnvelopeO
max_output_tokens: options.maxOutputTokens,
truncation: options.truncation ?? "auto",
...(options.temperature === undefined ? {} : { temperature: options.temperature }),
...(options.promptCacheKey ? { prompt_cache_key: options.promptCacheKey } : {}),
stream: true,
...(options.thinkingPayload ?? {}),
...(tools.length > 0 ? { tools, tool_choice: options.toolChoice } : {}),
Expand Down
47 changes: 47 additions & 0 deletions src/test/headers.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
import assert from "node:assert/strict";
import { describe, it, before } from "node:test";
import Module from "node:module";
import path from "node:path";
import fs from "node:fs";
import os from "node:os";

const vscodeMockPath = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "vscode-mock-headers-")), "index.js");
fs.writeFileSync(
vscodeMockPath,
`"use strict";
class LanguageModelChatToolMode { static Required = "required"; }
module.exports = { LanguageModelChatToolMode, workspace: { workspaceFolders: undefined } };
`,
"utf-8",
);

type ResolveFilename = (request: string, parent: unknown, ...args: unknown[]) => string;
const moduleResolver = Module as unknown as { _resolveFilename: ResolveFilename };
const originalResolveFilename = moduleResolver._resolveFilename;
moduleResolver._resolveFilename = function (request: string, parent: unknown, ...args: unknown[]): string {
if (request === "vscode") return vscodeMockPath;
return originalResolveFilename.call(this, request, parent, ...args);
};

let hashRawCacheKey: typeof import("../request/headers.js").hashRawCacheKey;

describe("hashRawCacheKey", () => {
before(async () => {
const mod = await import("../request/headers.js");
hashRawCacheKey = mod.hashRawCacheKey;
});

it("pins SHA256 for a fixed raw key (regression guard)", () => {
// Mirrors docs/references/opencode-context-cache-reference.md
const raw = "testuser@testhost:C:/project";
const expected = "4f77e704edc190c1872f5ac84f42320d082a4cf4c52bfdacd2634db07de24120";
assert.equal(hashRawCacheKey(raw), expected);
});

it("is stable for backslash vs forward-slash after normalization", () => {
const canonical = "testuser@testhost:C:/a/b";
const expected = hashRawCacheKey(canonical);
assert.equal(expected.length, 64);
assert.match(expected, /^[a-f0-9]{64}$/);
});
});
Loading