Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
62 changes: 41 additions & 21 deletions src/request/anthropic.ts
Original file line number Diff line number Diff line change
Expand Up @@ -55,25 +55,39 @@ export function buildAnthropicMessagesRequestBody(
}

export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequestMessage[] {
let cacheControlCount = 0;
const nextCacheControl = (): { cache_control?: AnthropicCacheControl } => {
cacheControlCount += 1;
return cacheControlCount <= 4 ? { cache_control: { type: "ephemeral" } } : {};
// Cache-stable breakpoints: at most 2 marks at structural boundaries —
// (1) end of the first (anchor/static-prefix) message and (2) end of the
// message before the latest user turn (growing-tail checkpoint). Per-block
// marking (first-4-blocks) shifted every later breakpoint whenever any
// earlier turn gained a text/image part, and exhausted the 4-slot budget
// that gateways may also inject into. Writes happen only at breakpoints and
// reads walk back max 20 blocks, so these two positions cover both the
// static prefix and the conversation tail without drift.
const markTargets = new Set<number>();
if (messages.length >= 1) markTargets.add(0);
if (messages.length >= 3) markTargets.add(messages.length - 2);
let messageIndex = 0;
const nextCacheControlFor = (msgIdx: number, isLastBlockOfMessage: boolean): { cache_control?: AnthropicCacheControl } => {
if (isLastBlockOfMessage && markTargets.has(msgIdx)) {
return { cache_control: { type: "ephemeral" } };
}
return {};
};

const anthropicMessages: AnthropicRequestMessage[] = [];

for (const message of messages) {
const msgIdx = messageIndex++;
if (message.role === "user") {
const userBlocks = anthropicUserBlocks(message.content, nextCacheControl);
const userBlocks = anthropicUserBlocks(message.content, (last) => nextCacheControlFor(msgIdx, last));
if (userBlocks.length) {
anthropicMessages.push({ role: "user", content: userBlocks });
}
continue;
}

if (message.role === "assistant") {
const assistantBlocks = anthropicAssistantBlocks(message, nextCacheControl);
const assistantBlocks = anthropicAssistantBlocks(message, (last) => nextCacheControlFor(msgIdx, last));
if (assistantBlocks.length) {
anthropicMessages.push({ role: "assistant", content: assistantBlocks });
}
Expand All @@ -88,8 +102,8 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest
{
type: "tool_result",
tool_use_id: message.tool_call_id,
content: anthropicToolResultContent(message.content, nextCacheControl),
...nextCacheControl(),
content: anthropicToolResultContent(message.content),
...nextCacheControlFor(msgIdx, true),
},
],
});
Expand All @@ -99,7 +113,7 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest
if (!anthropicMessages.length) {
anthropicMessages.push({
role: "user",
content: [{ type: "text", text: "Continue the conversation.", ...nextCacheControl() }],
content: [{ type: "text", text: "Continue the conversation." }],
});
}

Expand All @@ -108,30 +122,36 @@ export function buildAnthropicMessages(messages: ApiMessage[]): AnthropicRequest

function anthropicUserBlocks(
content: ApiMessage["content"],
nextCacheControl: () => { cache_control?: AnthropicCacheControl },
mark: (isLastBlockOfMessage: boolean) => { cache_control?: AnthropicCacheControl },
): AnthropicContentBlock[] {
if (typeof content === "string") {
return content.trim() ? [{ type: "text", text: content, ...nextCacheControl() }] : [];
return content.trim() ? [{ type: "text", text: content, ...mark(true) }] : [];
}

if (!Array.isArray(content)) {
return [];
}

const blocks: AnthropicContentBlock[] = [];
// Collect candidate blocks first so only the LAST block of the message
// carries the breakpoint — earlier blocks stay unmarked and stable.
const raw: AnthropicContentBlock[] = [];
for (const part of content) {
if (part.type === "text" && typeof part.text === "string" && part.text.length > 0) {
blocks.push({ type: "text", text: part.text, ...nextCacheControl() });
raw.push({ type: "text", text: part.text });
continue;
}

if (part.type === "image_url") {
const source = anthropicImageSource(part);
if (source) {
blocks.push({ type: "image", source, ...nextCacheControl() });
raw.push({ type: "image", source });
}
}
}
raw.forEach((block, i) => {
blocks.push(i === raw.length - 1 ? { ...block, ...mark(true) } : block);
});

return blocks;
}
Expand All @@ -142,10 +162,7 @@ function anthropicUserBlocks(
// (text + image blocks) only when an image_url part is present. This keeps
// text-only tool results byte-for-byte identical to the previous behavior
// while enabling vision-capable Anthropic models to consume MCP screenshots.
function anthropicToolResultContent(
content: ApiMessage["content"],
nextCacheControl: () => { cache_control?: AnthropicCacheControl },
): string | AnthropicContentBlock[] {
function anthropicToolResultContent(content: ApiMessage["content"]): string | AnthropicContentBlock[] {
if (typeof content === "string") {
return content;
}
Expand All @@ -159,18 +176,18 @@ function anthropicToolResultContent(
return joinedTextContent(content, "\n");
}

return anthropicUserBlocks(content, nextCacheControl);
return anthropicUserBlocks(content, () => ({}));
}

function anthropicAssistantBlocks(
message: ApiMessage,
nextCacheControl: () => { cache_control?: AnthropicCacheControl },
mark: (isLastBlockOfMessage: boolean) => { cache_control?: AnthropicCacheControl },
): AnthropicContentBlock[] {
const blocks: AnthropicContentBlock[] = [];

const text = joinedTextContent(message.content);
if (text) {
blocks.push({ type: "text", text, ...nextCacheControl() });
blocks.push({ type: "text", text });
}

for (const toolCall of message.tool_calls ?? []) {
Expand All @@ -179,9 +196,12 @@ function anthropicAssistantBlocks(
id: toolCall.id || `toolu_${Math.random().toString(36).slice(2)}`,
name: toolCall.function.name,
input: anthropicToolCallInput(toolCall.function.arguments),
...nextCacheControl(),
});
}
if (blocks.length) {
const last = blocks.length - 1;
blocks[last] = { ...blocks[last], ...mark(true) };
}

return blocks;
}
Expand Down
78 changes: 60 additions & 18 deletions src/request/headers.ts
Original file line number Diff line number Diff line change
Expand Up @@ -98,29 +98,51 @@ export function resolveProjectCacheKey(modelId: string): string | null {
// gateway reads x-opencode-session first, then converts that sticky identifier
// into provider-specific affinity headers such as x-session-affinity upstream.
//
// VS Code's provider API does not currently expose a guaranteed public session
// identifier everywhere, so we first probe a few known internal fields and then
// fall back to a stable hash of the first messages in the conversation. That
// preserves sticky routing and cache affinity without depending on hidden state.
// VS Code's provider API does not expose a stable chat-session identity
// (microsoft/vscode#305853 — `chatSessionResource` proposed, not shipped), so
// we probe known internal option fields and fall back to a memoized anchor
// hash. The anchor covers ONLY the first message + sorted tool names + model:
// history trimming rewrites messages[1..N] by design, so hashing the first 3
// messages would shift the session id on every trim and break sticky routing
// + prefix-cache affinity. Memoization keeps the id constant for the life of
// the conversation even as the tail grows.
const sessionMemo = new Map<string, string>();

export function buildOpenCodeRequestHeaders(
messages: readonly vscode.LanguageModelChatRequestMessage[],
options: vscode.ProvideLanguageModelChatResponseOptions,
modelId: string,
): Record<string, string> {
const sessionId = cleanHeaderValue(
findStringOption(options, [
"sessionId",
"sessionID",
"chatSessionId",
"chatSessionID",
"conversationId",
"conversationID",
"threadId",
"threadID",
"session.id",
"chatSession.id",
]) ?? `vscode-${stableHash(conversationAnchor(messages, modelId))}`,
);
const explicit = findStringOption(options, [
"sessionId",
"sessionID",
"chatSessionId",
"chatSessionID",
"conversationId",
"conversationID",
"threadId",
"threadID",
"session.id",
"chatSession.id",
]);
let sessionId: string;
if (explicit) {
sessionId = cleanHeaderValue(explicit);
} else {
const anchorKey = stableConversationAnchorKey(messages, options, modelId);
const memoKey = `${modelId}:${anchorKey}`;
const memoized = sessionMemo.get(memoKey);
if (memoized) {
sessionId = memoized;
} else {
sessionId = cleanHeaderValue(`vscode-${anchorKey}`);
sessionMemo.set(memoKey, sessionId);
if (sessionMemo.size > 500) {
const oldest = sessionMemo.keys().next().value;
if (oldest) sessionMemo.delete(oldest);
}
}
}
const requestId = cleanHeaderValue(
findStringOption(options, ["requestId", "requestID", "messageId", "messageID"]) ??
`req-${stableHash(`${String(Date.now())}-${String(Math.random())}-${sessionId}-${modelId}`)}`,
Expand Down Expand Up @@ -188,10 +210,30 @@ export function readPath(value: unknown, path: string[]): unknown {
}

export function conversationAnchor(messages: readonly vscode.LanguageModelChatRequestMessage[], modelId: string): string {
// Legacy anchor (first-3-messages) kept for compatibility; new code should
// use stableConversationAnchorKey which is trim-stable.
const anchorMessages = messages.slice(0, 3).map((message) => `${String(message.role)}:${messageText(message).slice(0, 2048)}`);
return anchorMessages.length ? anchorMessages.join("\n") : modelId;
}

/**
* Trim-stable conversation key: first message text + sorted tool names +
* model id. Messages[1..] shift on every history trim by design and must not
* feed the session hash (see buildOpenCodeRequestHeaders).
*/
export function stableConversationAnchorKey(
messages: readonly vscode.LanguageModelChatRequestMessage[],
options: vscode.ProvideLanguageModelChatResponseOptions,
modelId: string,
): string {
const first = messages.length ? `${String(messages[0].role)}:${messageText(messages[0]).slice(0, 2048)}` : modelId;
const toolNames = [...(options.tools ?? [])]
.map((t) => t.name)
.sort()
.join(",");
return stableHash(`${first}\ntools:${toolNames}\nmodel:${modelId}`);
}

export function cleanHeaderValue(value: string): string {
const cleaned = value.replace(/[\r\n]/g, " ").trim();
return cleaned ? cleaned.slice(0, 256) : "unknown";
Expand Down
17 changes: 11 additions & 6 deletions src/responsesRequest.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
import { randomUUID } from "node:crypto";
import { createHash } from "node:crypto";

export interface ResponsesRequestEnvelopeOptions {
model: string;
Expand Down Expand Up @@ -186,13 +186,18 @@ export function pairResponsesFunctionCallItems(items: Record<string, unknown>[])

/**
* Return a Responses-API-compliant `function_call` item id. Ids already in the
* `fc_` namespace pass through unchanged; anything else (chat-completions
* `call_*` ids, gateway ids, empty strings) is replaced with a fresh `fc_`
* synthetic id. Deterministic per call site is not required — the id only has
* to be valid and unique within the request.
* `fc_` namespace pass through unchanged (verbatim echo preserves the
* provider's prefix cache); anything else (chat-completions `call_*` ids,
* gateway ids, empty strings) maps deterministically to `fc_<sha256>` so the
* same history replays byte-identically on every turn. A random id per request
* would break the Responses prefix cache at the first tool call on every turn
* (cf. OpenHands SDK #2905). The `call_id` correlation id is always preserved
* verbatim separately — only this item `id` is namespaced.
*/
export function responsesFunctionCallItemId(originalId: string): string {
return originalId.startsWith("fc_") ? originalId : `fc_${randomUUID().replace(/-/g, "")}`;
if (originalId.startsWith("fc_")) return originalId;
const digest = createHash("sha256").update(originalId, "utf8").digest("hex").slice(0, 32);
return `fc_${digest}`;
}

/** Narrow a union value to a content-part array without falling back to `any[]`. */
Expand Down
22 changes: 22 additions & 0 deletions src/test/responsesRequest.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -199,3 +199,25 @@ describe("pairResponsesFunctionCallItems (issue #216)", () => {
assert.equal((paired[1] as { output: string }).output, "first");
});
});

describe("responsesFunctionCallItemId (prefix-cache stability)", () => {
it("is deterministic for non-fc ids across turns", async () => {
const { responsesFunctionCallItemId } = await import("../responsesRequest.js");
const first = responsesFunctionCallItemId("call_abc123");
const second = responsesFunctionCallItemId("call_abc123");
assert.equal(first, second);
assert.match(first, /^fc_[0-9a-f]{32}$/);
});

it("replays a 3-turn trace byte-identically (no random ids)", async () => {
const { responsesInputItemsFromMessage } = await import("../responsesRequest.js");
const history = [
{ role: "assistant", content: null, tool_calls: [{ id: "call_1", type: "function", function: { name: "f", arguments: "{}" } }] },
{ role: "tool", tool_call_id: "call_1", content: "ok" },
] as never[];
const once = responsesInputItemsFromMessage(history[0]);
const twice = responsesInputItemsFromMessage(history[0]);
assert.equal(JSON.stringify(once), JSON.stringify(twice));
assert.equal((once[0] as { id: string }).id, (twice[0] as { id: string }).id);
});
});
Loading