diff --git a/src/provider/chatPrep.ts b/src/provider/chatPrep.ts index f58407477..9c6d71a7b 100644 --- a/src/provider/chatPrep.ts +++ b/src/provider/chatPrep.ts @@ -27,6 +27,7 @@ import { modelLimits, resolveRawModelId, } from "./settings"; +import { createHash } from "node:crypto"; import { messagesHaveImages } from "../request/builders"; import { buildOpenCodeRequestHeaders } from "../request/headers"; import type { ApiMessage, ApiSettings, OpenAiContentPart } from "../request/types"; @@ -281,6 +282,29 @@ export async function prepareChatRequest( // call), so creating one per request flooded the Output tab with dozens of // duplicate "OpenCode" channels (issue #220). Transports receive the // provider's shared channel instead. + { + const flag = process.env.OPENCODE_CONTEXT_CACHE_DEBUG ?? ""; + const debugEnabled = flag === "1" || flag === "true"; + if (debugEnabled) { + try { + const prefixAnchor = apiMessages + .slice(0, 3) + .map((m) => `${m.role}:${JSON.stringify(m.content).slice(0, 2048)}`) + .join("\n"); + const toolNames = [...(options.tools ?? [])] + .map((t) => (t as { name: string }).name) + .sort((a, b) => (a < b ? -1 : a > b ? 1 : 0)) + .join(","); + const prefixHash = createHash("sha256") + .update(prefixAnchor + "|" + toolNames, "utf8") + .digest("hex") + .slice(0, 12); + deps.log(`[prefix-hash] ${prefixHash} messages=${String(apiMessages.length)} tools=${String(options.tools?.length ?? 0)}`); + } catch (e) { + deps.log(`[prefix-hash] failed: ${String(e)}`); + } + } + } const onTransportSummary = (summary: TransportRequestSummary) => { // Compute credits for VS Code session cost (1 credit = $0.01). // VS Code reads usage.copilotCredits from the LanguageModelDataPart diff --git a/src/request/anthropic.ts b/src/request/anthropic.ts index 730473ba2..593bffae8 100644 --- a/src/request/anthropic.ts +++ b/src/request/anthropic.ts @@ -221,7 +221,8 @@ function anthropicImageSource(part: OpenAiContentPart): AnthropicImageSource | u } function mapAnthropicTools(tools: readonly vscode.LanguageModelChatTool[] | undefined): AnthropicToolDefinition[] { - return (tools ?? []).map((tool) => ({ + const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); + return sorted.map((tool) => ({ name: tool.name, description: tool.description, input_schema: sanitizeToolSchema(tool.inputSchema), diff --git a/src/request/google.ts b/src/request/google.ts index c498afe9a..1c5dfe9f6 100644 --- a/src/request/google.ts +++ b/src/request/google.ts @@ -33,7 +33,8 @@ export function buildGoogleGenerateContentBody( } function mapGoogleTools(tools: readonly vscode.LanguageModelChatTool[] | undefined): Record[] { - return (tools ?? []).map((tool) => ({ + const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); + return sorted.map((tool) => ({ name: tool.name, description: tool.description, parameters: sanitizeToolSchema(tool.inputSchema), diff --git a/src/request/headers.ts b/src/request/headers.ts index 6135a623e..5bb9e8c03 100644 --- a/src/request/headers.ts +++ b/src/request/headers.ts @@ -1,5 +1,8 @@ import * as vscode from "vscode"; -import { randomUUID } from "node:crypto"; +import { createHash, randomUUID } from "node:crypto"; +import * as os from "os"; +import * as path from "path"; +import * as fs from "fs"; import { OPEN_CODE_CLIENT } from "../config"; import { getUserAgent } from "../provider/definitions"; import { messageText } from "../provider/tokens"; @@ -30,6 +33,67 @@ export function auxiliarySessionId(context: vscode.ExtensionContext): string { return id; } +// --- Context-cache parity (mirrors ~/.config/opencode/plugins/opencode-context-cache.mjs) --- +// NOTE (PR #212 review): the legacy model headers x-session-id / +// conversation_id / session_id were dropped — no evidence they do anything +// beyond x-opencode-session + prompt_cache_key, and they are not in any +// public Zen docs. The project cache key now flows only via prompt_cache_key +// (chat-completions/responses bodies); x-opencode-session stays the routing +// affinity header. +const CONTEXT_CACHE_DEBUG_ENV_VAR = "OPENCODE_CONTEXT_CACHE_DEBUG"; + +function appendContextCacheLog(message: string): void { + const flag = process.env[CONTEXT_CACHE_DEBUG_ENV_VAR] ?? ""; + if (flag !== "1" && flag !== "true") return; + try { + const logPath = path.join(os.homedir(), ".config", "opencode", "plugins", "context-cache-vscode.log"); + const safe = message.replace(/\n/g, "\\n").replace(/\r/g, "\\r"); + const line = `[${new Date().toISOString()}] [pid:${String(process.pid)}] [context-cache-vscode] ${safe}\n`; + fs.appendFileSync(logPath, line, "utf8"); + } catch { + /* best-effort */ + } +} + +export function hashRawCacheKey(raw: string): string { + return createHash("sha256").update(raw, "utf8").digest("hex"); +} + +function normalizeDirForCacheKey(dir: string): string { + // Canonicalize separators so C:\a\b and C:/a/b hash identically. + // Drive-letter upper-casing keeps c:\ vs C:\ stable on Windows. + let out = dir.replace(/\\/g, "/"); + if (out.length >= 2 && out[1] === ":" && out[0] !== out[0].toUpperCase()) out = out[0].toUpperCase() + out.slice(1); + return out; +} + +function resolveRawProjectCacheKey(modelId: string): string | null { + const env = process.env as Record; + const override = (env.OPENCODE_PROMPT_CACHE_KEY ?? env.OPENCODE_STICKY_SESSION_ID ?? "").trim(); + if (override) return override; + try { + const user = env.USERNAME ?? env.USER ?? env.LOGNAME ?? "unknown"; + const host = os.hostname(); + let dir = ""; + try { + dir = vscode.workspace.workspaceFolders?.[0]?.uri.fsPath ?? ""; + } catch { + dir = ""; + } + if (dir) dir = normalizeDirForCacheKey(dir); + if (!dir) dir = modelId || "no-workspace"; + return `${user}@${host}:${dir}`; + } catch { + return null; + } +} + +export function resolveProjectCacheKey(modelId: string): string | null { + const raw = resolveRawProjectCacheKey(modelId); + if (!raw) return null; + return hashRawCacheKey(raw); +} + // The official OpenCode client sends these headers on every request. The Zen // gateway reads x-opencode-session first, then converts that sticky identifier // into provider-specific affinity headers such as x-session-affinity upstream. @@ -62,12 +126,19 @@ export function buildOpenCodeRequestHeaders( `req-${stableHash(`${String(Date.now())}-${String(Math.random())}-${sessionId}-${modelId}`)}`, ); - return { + const projectCacheKey = resolveProjectCacheKey(modelId); + const headers: Record = { "x-opencode-session": sessionId, "x-opencode-request": requestId, "x-opencode-client": OPEN_CODE_CLIENT, "User-Agent": getUserAgent(), }; + if (projectCacheKey) { + appendContextCacheLog(`model=${modelId} raw=${resolveRawProjectCacheKey(modelId) ?? ""} hash=${projectCacheKey}`); + } else { + appendContextCacheLog(`model=${modelId} no stable cache key resolved`); + } + return headers; } /** diff --git a/src/request/openai.ts b/src/request/openai.ts index 964cfdc8f..1eaff96a9 100644 --- a/src/request/openai.ts +++ b/src/request/openai.ts @@ -12,6 +12,7 @@ import { lookupModelRegistryEntry } from "../core/registry"; import { buildResponsesRequestEnvelope, pairResponsesFunctionCallItems, responsesInputItemsFromMessage } from "../responsesRequest"; import { thinkingProviderFor } from "../thinking"; import { sanitizeToolSchema } from "./schema"; +import { resolveProjectCacheKey } from "./headers"; import { messagesHaveImages } from "./shared"; import type { ResolvedModelMetadata } from "../models/metadata"; import type { ModelLimits } from "../models/modelLimits"; @@ -39,6 +40,10 @@ export function buildChatCompletionsRequestBody( max_tokens: limits.maxOutputTokens, stream: true, stream_options: { include_usage: true }, + ...(() => { + const k = resolveProjectCacheKey(modelId); + return k ? { prompt_cache_key: k } : {}; + })(), ...thinkingPayload, ...(tools.length ? { tools, tool_choice: toolChoice(options.toolMode) } : {}), }; @@ -66,6 +71,10 @@ export function buildResponsesRequestBody( model: modelId, input, maxOutputTokens: limits.maxOutputTokens, + ...(() => { + const k = resolveProjectCacheKey(modelId); + return k ? { promptCacheKey: k } : {}; + })(), // Muse Spark gateway requires `truncation: "disabled"` — requests whose // input exceeds the 1M context window will hard-fail (HTTP 400) instead // of being silently truncated upstream. This is a gateway constraint, @@ -80,7 +89,8 @@ export function buildResponsesRequestBody( } function mapOpenAiTools(tools: readonly vscode.LanguageModelChatTool[] | undefined): OpenAiToolDefinition[] { - return (tools ?? []).map((tool) => ({ + const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); + return sorted.map((tool) => ({ type: "function", function: { name: tool.name, @@ -92,7 +102,8 @@ function mapOpenAiTools(tools: readonly vscode.LanguageModelChatTool[] | undefin function mapResponsesTools(tools: readonly vscode.LanguageModelChatTool[] | undefined, modelId?: string): Record[] { const needsTruncation = isMuseFamily(modelId ?? ""); - return (tools ?? []).map((tool) => ({ + const sorted = [...(tools ?? [])].sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); + return sorted.map((tool) => ({ type: "function", name: needsTruncation ? truncateToolName(tool.name) : tool.name, description: tool.description, diff --git a/src/request/schema.ts b/src/request/schema.ts index 9a9f2fafa..e5fc4e7d8 100644 --- a/src/request/schema.ts +++ b/src/request/schema.ts @@ -61,7 +61,7 @@ function sanitizeJsonSchemaNode(value: unknown, root: Record, s } const result: Record = {}; - for (const [key, child] of Object.entries(value)) { + for (const [key, child] of Object.entries(value).sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) { if (key === "$schema" || key === "$id" || key === "$ref" || key === "$defs" || key === "definitions") { continue; } diff --git a/src/responsesRequest.ts b/src/responsesRequest.ts index 25647fdef..3d47f55cf 100644 --- a/src/responsesRequest.ts +++ b/src/responsesRequest.ts @@ -9,6 +9,7 @@ export interface ResponsesRequestEnvelopeOptions { tools?: readonly unknown[]; toolChoice?: unknown; truncation?: "auto" | "disabled"; + promptCacheKey?: string; } /** @@ -46,6 +47,7 @@ export function buildResponsesRequestEnvelope(options: ResponsesRequestEnvelopeO max_output_tokens: options.maxOutputTokens, truncation: options.truncation ?? "auto", ...(options.temperature === undefined ? {} : { temperature: options.temperature }), + ...(options.promptCacheKey ? { prompt_cache_key: options.promptCacheKey } : {}), stream: true, ...(options.thinkingPayload ?? {}), ...(tools.length > 0 ? { tools, tool_choice: options.toolChoice } : {}), diff --git a/src/test/headers.test.ts b/src/test/headers.test.ts new file mode 100644 index 000000000..c3da1a38f --- /dev/null +++ b/src/test/headers.test.ts @@ -0,0 +1,47 @@ +import assert from "node:assert/strict"; +import { describe, it, before } from "node:test"; +import Module from "node:module"; +import path from "node:path"; +import fs from "node:fs"; +import os from "node:os"; + +const vscodeMockPath = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "vscode-mock-headers-")), "index.js"); +fs.writeFileSync( + vscodeMockPath, + `"use strict"; +class LanguageModelChatToolMode { static Required = "required"; } +module.exports = { LanguageModelChatToolMode, workspace: { workspaceFolders: undefined } }; +`, + "utf-8", +); + +type ResolveFilename = (request: string, parent: unknown, ...args: unknown[]) => string; +const moduleResolver = Module as unknown as { _resolveFilename: ResolveFilename }; +const originalResolveFilename = moduleResolver._resolveFilename; +moduleResolver._resolveFilename = function (request: string, parent: unknown, ...args: unknown[]): string { + if (request === "vscode") return vscodeMockPath; + return originalResolveFilename.call(this, request, parent, ...args); +}; + +let hashRawCacheKey: typeof import("../request/headers.js").hashRawCacheKey; + +describe("hashRawCacheKey", () => { + before(async () => { + const mod = await import("../request/headers.js"); + hashRawCacheKey = mod.hashRawCacheKey; + }); + + it("pins SHA256 for a fixed raw key (regression guard)", () => { + // Mirrors docs/references/opencode-context-cache-reference.md + const raw = "testuser@testhost:C:/project"; + const expected = "4f77e704edc190c1872f5ac84f42320d082a4cf4c52bfdacd2634db07de24120"; + assert.equal(hashRawCacheKey(raw), expected); + }); + + it("is stable for backslash vs forward-slash after normalization", () => { + const canonical = "testuser@testhost:C:/a/b"; + const expected = hashRawCacheKey(canonical); + assert.equal(expected.length, 64); + assert.match(expected, /^[a-f0-9]{64}$/); + }); +});