diff --git a/src/request/anthropic.ts b/src/request/anthropic.ts index 593bffae8..b9ac7078b 100644 --- a/src/request/anthropic.ts +++ b/src/request/anthropic.ts @@ -55,17 +55,31 @@ export function buildAnthropicMessagesRequestBody( } export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequestMessage[] { - let cacheControlCount = 0; - const nextCacheControl = (): { cache_control?: AnthropicCacheControl } => { - cacheControlCount += 1; - return cacheControlCount <= 4 ? { cache_control: { type: "ephemeral" } } : {}; + // Cache-stable breakpoints: at most 2 marks at structural boundaries — + // (1) end of the first (anchor/static-prefix) message and (2) end of the + // message before the latest user turn (growing-tail checkpoint). Per-block + // marking (first-4-blocks) shifted every later breakpoint whenever any + // earlier turn gained a text/image part, and exhausted the 4-slot budget + // that gateways may also inject into. Writes happen only at breakpoints and + // reads walk back max 20 blocks, so these two positions cover both the + // static prefix and the conversation tail without drift. + const markTargets = new Set(); + if (messages.length >= 1) markTargets.add(0); + if (messages.length >= 3) markTargets.add(messages.length - 2); + let messageIndex = 0; + const nextCacheControlFor = (msgIdx: number, isLastBlockOfMessage: boolean): { cache_control?: AnthropicCacheControl } => { + if (isLastBlockOfMessage && markTargets.has(msgIdx)) { + return { cache_control: { type: "ephemeral" } }; + } + return {}; }; const anthropicMessages: AnthropicRequestMessage[] = []; for (const message of messages) { + const msgIdx = messageIndex++; if (message.role === "user") { - const userBlocks = anthropicUserBlocks(message.content, nextCacheControl); + const userBlocks = anthropicUserBlocks(message.content, (last) => nextCacheControlFor(msgIdx, last)); if (userBlocks.length) { anthropicMessages.push({ role: "user", content: userBlocks }); } @@ -73,7 +87,7 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest } if (message.role === "assistant") { - const assistantBlocks = anthropicAssistantBlocks(message, nextCacheControl); + const assistantBlocks = anthropicAssistantBlocks(message, (last) => nextCacheControlFor(msgIdx, last)); if (assistantBlocks.length) { anthropicMessages.push({ role: "assistant", content: assistantBlocks }); } @@ -88,8 +102,8 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest { type: "tool_result", tool_use_id: message.tool_call_id, - content: anthropicToolResultContent(message.content, nextCacheControl), - ...nextCacheControl(), + content: anthropicToolResultContent(message.content), + ...nextCacheControlFor(msgIdx, true), }, ], }); @@ -99,7 +113,7 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest if (!anthropicMessages.length) { anthropicMessages.push({ role: "user", - content: [{ type: "text", text: "Continue the conversation.", ...nextCacheControl() }], + content: [{ type: "text", text: "Continue the conversation." }], }); } @@ -108,10 +122,10 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest function anthropicUserBlocks( content: ApiMessage["content"], - nextCacheControl: () => { cache_control?: AnthropicCacheControl }, + mark: (isLastBlockOfMessage: boolean) => { cache_control?: AnthropicCacheControl }, ): AnthropicContentBlock[] { if (typeof content === "string") { - return content.trim() ? [{ type: "text", text: content, ...nextCacheControl() }] : []; + return content.trim() ? [{ type: "text", text: content, ...mark(true) }] : []; } if (!Array.isArray(content)) { @@ -119,19 +133,25 @@ function anthropicUserBlocks( } const blocks: AnthropicContentBlock[] = []; + // Collect candidate blocks first so only the LAST block of the message + // carries the breakpoint — earlier blocks stay unmarked and stable. + const raw: AnthropicContentBlock[] = []; for (const part of content) { if (part.type === "text" && typeof part.text === "string" && part.text.length > 0) { - blocks.push({ type: "text", text: part.text, ...nextCacheControl() }); + raw.push({ type: "text", text: part.text }); continue; } if (part.type === "image_url") { const source = anthropicImageSource(part); if (source) { - blocks.push({ type: "image", source, ...nextCacheControl() }); + raw.push({ type: "image", source }); } } } + raw.forEach((block, i) => { + blocks.push(i === raw.length - 1 ? { ...block, ...mark(true) } : block); + }); return blocks; } @@ -142,10 +162,7 @@ function anthropicUserBlocks( // (text + image blocks) only when an image_url part is present. This keeps // text-only tool results byte-for-byte identical to the previous behavior // while enabling vision-capable Anthropic models to consume MCP screenshots. -function anthropicToolResultContent( - content: ApiMessage["content"], - nextCacheControl: () => { cache_control?: AnthropicCacheControl }, -): string | AnthropicContentBlock[] { +function anthropicToolResultContent(content: ApiMessage["content"]): string | AnthropicContentBlock[] { if (typeof content === "string") { return content; } @@ -159,18 +176,18 @@ function anthropicToolResultContent( return joinedTextContent(content, "\n"); } - return anthropicUserBlocks(content, nextCacheControl); + return anthropicUserBlocks(content, () => ({})); } function anthropicAssistantBlocks( message: ApiMessage, - nextCacheControl: () => { cache_control?: AnthropicCacheControl }, + mark: (isLastBlockOfMessage: boolean) => { cache_control?: AnthropicCacheControl }, ): AnthropicContentBlock[] { const blocks: AnthropicContentBlock[] = []; const text = joinedTextContent(message.content); if (text) { - blocks.push({ type: "text", text, ...nextCacheControl() }); + blocks.push({ type: "text", text }); } for (const toolCall of message.tool_calls ?? []) { @@ -179,9 +196,12 @@ function anthropicAssistantBlocks( id: toolCall.id || `toolu_${Math.random().toString(36).slice(2)}`, name: toolCall.function.name, input: anthropicToolCallInput(toolCall.function.arguments), - ...nextCacheControl(), }); } + if (blocks.length) { + const last = blocks.length - 1; + blocks[last] = { ...blocks[last], ...mark(true) }; + } return blocks; } diff --git a/src/request/headers.ts b/src/request/headers.ts index 5bb9e8c03..e0c7bb6f1 100644 --- a/src/request/headers.ts +++ b/src/request/headers.ts @@ -98,29 +98,51 @@ export function resolveProjectCacheKey(modelId: string): string | null { // gateway reads x-opencode-session first, then converts that sticky identifier // into provider-specific affinity headers such as x-session-affinity upstream. // -// VS Code's provider API does not currently expose a guaranteed public session -// identifier everywhere, so we first probe a few known internal fields and then -// fall back to a stable hash of the first messages in the conversation. That -// preserves sticky routing and cache affinity without depending on hidden state. +// VS Code's provider API does not expose a stable chat-session identity +// (microsoft/vscode#305853 — `chatSessionResource` proposed, not shipped), so +// we probe known internal option fields and fall back to a memoized anchor +// hash. The anchor covers ONLY the first message + sorted tool names + model: +// history trimming rewrites messages[1..N] by design, so hashing the first 3 +// messages would shift the session id on every trim and break sticky routing +// + prefix-cache affinity. Memoization keeps the id constant for the life of +// the conversation even as the tail grows. +const sessionMemo = new Map(); + export function buildOpenCodeRequestHeaders( messages: readonly vscode.LanguageModelChatRequestMessage[], options: vscode.ProvideLanguageModelChatResponseOptions, modelId: string, ): Record { - const sessionId = cleanHeaderValue( - findStringOption(options, [ - "sessionId", - "sessionID", - "chatSessionId", - "chatSessionID", - "conversationId", - "conversationID", - "threadId", - "threadID", - "session.id", - "chatSession.id", - ]) ?? `vscode-${stableHash(conversationAnchor(messages, modelId))}`, - ); + const explicit = findStringOption(options, [ + "sessionId", + "sessionID", + "chatSessionId", + "chatSessionID", + "conversationId", + "conversationID", + "threadId", + "threadID", + "session.id", + "chatSession.id", + ]); + let sessionId: string; + if (explicit) { + sessionId = cleanHeaderValue(explicit); + } else { + const anchorKey = stableConversationAnchorKey(messages, options, modelId); + const memoKey = `${modelId}:${anchorKey}`; + const memoized = sessionMemo.get(memoKey); + if (memoized) { + sessionId = memoized; + } else { + sessionId = cleanHeaderValue(`vscode-${anchorKey}`); + sessionMemo.set(memoKey, sessionId); + if (sessionMemo.size > 500) { + const oldest = sessionMemo.keys().next().value; + if (oldest) sessionMemo.delete(oldest); + } + } + } const requestId = cleanHeaderValue( findStringOption(options, ["requestId", "requestID", "messageId", "messageID"]) ?? `req-${stableHash(`${String(Date.now())}-${String(Math.random())}-${sessionId}-${modelId}`)}`, @@ -188,10 +210,30 @@ export function readPath(value: unknown, path: string[]): unknown { } export function conversationAnchor(messages: readonly vscode.LanguageModelChatRequestMessage[], modelId: string): string { + // Legacy anchor (first-3-messages) kept for compatibility; new code should + // use stableConversationAnchorKey which is trim-stable. const anchorMessages = messages.slice(0, 3).map((message) => `${String(message.role)}:${messageText(message).slice(0, 2048)}`); return anchorMessages.length ? anchorMessages.join("\n") : modelId; } +/** + * Trim-stable conversation key: first message text + sorted tool names + + * model id. Messages[1..] shift on every history trim by design and must not + * feed the session hash (see buildOpenCodeRequestHeaders). + */ +export function stableConversationAnchorKey( + messages: readonly vscode.LanguageModelChatRequestMessage[], + options: vscode.ProvideLanguageModelChatResponseOptions, + modelId: string, +): string { + const first = messages.length ? `${String(messages[0].role)}:${messageText(messages[0]).slice(0, 2048)}` : modelId; + const toolNames = [...(options.tools ?? [])] + .map((t) => t.name) + .sort() + .join(","); + return stableHash(`${first}\ntools:${toolNames}\nmodel:${modelId}`); +} + export function cleanHeaderValue(value: string): string { const cleaned = value.replace(/[\r\n]/g, " ").trim(); return cleaned ? cleaned.slice(0, 256) : "unknown"; diff --git a/src/responsesRequest.ts b/src/responsesRequest.ts index 3e546170f..e0b89319f 100644 --- a/src/responsesRequest.ts +++ b/src/responsesRequest.ts @@ -1,4 +1,4 @@ -import { randomUUID } from "node:crypto"; +import { createHash } from "node:crypto"; export interface ResponsesRequestEnvelopeOptions { model: string; @@ -186,13 +186,18 @@ export function pairResponsesFunctionCallItems(items: Record[]) /** * Return a Responses-API-compliant `function_call` item id. Ids already in the - * `fc_` namespace pass through unchanged; anything else (chat-completions - * `call_*` ids, gateway ids, empty strings) is replaced with a fresh `fc_` - * synthetic id. Deterministic per call site is not required — the id only has - * to be valid and unique within the request. + * `fc_` namespace pass through unchanged (verbatim echo preserves the + * provider's prefix cache); anything else (chat-completions `call_*` ids, + * gateway ids, empty strings) maps deterministically to `fc_` so the + * same history replays byte-identically on every turn. A random id per request + * would break the Responses prefix cache at the first tool call on every turn + * (cf. OpenHands SDK #2905). The `call_id` correlation id is always preserved + * verbatim separately — only this item `id` is namespaced. */ export function responsesFunctionCallItemId(originalId: string): string { - return originalId.startsWith("fc_") ? originalId : `fc_${randomUUID().replace(/-/g, "")}`; + if (originalId.startsWith("fc_")) return originalId; + const digest = createHash("sha256").update(originalId, "utf8").digest("hex").slice(0, 32); + return `fc_${digest}`; } /** Narrow a union value to a content-part array without falling back to `any[]`. */ diff --git a/src/test/responsesRequest.test.ts b/src/test/responsesRequest.test.ts index 7f4b8f079..6afe96941 100644 --- a/src/test/responsesRequest.test.ts +++ b/src/test/responsesRequest.test.ts @@ -199,3 +199,25 @@ describe("pairResponsesFunctionCallItems (issue #216)", () => { assert.equal((paired[1] as { output: string }).output, "first"); }); }); + +describe("responsesFunctionCallItemId (prefix-cache stability)", () => { + it("is deterministic for non-fc ids across turns", async () => { + const { responsesFunctionCallItemId } = await import("../responsesRequest.js"); + const first = responsesFunctionCallItemId("call_abc123"); + const second = responsesFunctionCallItemId("call_abc123"); + assert.equal(first, second); + assert.match(first, /^fc_[0-9a-f]{32}$/); + }); + + it("replays a 3-turn trace byte-identically (no random ids)", async () => { + const { responsesInputItemsFromMessage } = await import("../responsesRequest.js"); + const history = [ + { role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: "{}" } }] }, + { role: "tool", tool_call_id: "call_1", content: "ok" }, + ] as never[]; + const once = responsesInputItemsFromMessage(history[0]); + const twice = responsesInputItemsFromMessage(history[0]); + assert.equal(JSON.stringify(once), JSON.stringify(twice)); + assert.equal((once[0] as { id: string }).id, (twice[0] as { id: string }).id); + }); +});