diff --git a/CHANGELOG.md b/CHANGELOG.md index ead86ad7c..8c0819e0d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,26 @@ All notable changes to the **OpenCode Go BYOK Provider** extension are documented here. +## [0.7.5] — 2026-09-08 + +### Fixed + +- **`[Gateway]` `x-opencode-session` is now sent on every OpenCode request, not just chat.** OpenCode's Go docs (updated 2026-09-07) require a stable session id on all requests, and requests missing it may start erroring from 2026-09-06. Four auxiliary fetch sites hit the gateway without the header: `GET /models`, inline completions, `GET /usage` server sync, and the Manage-Provider test connection. All four now send a persisted per-installation session id; the main chat path keeps its per-conversation id. See `docs/issues/99-20260908-issue218-223-triage-dropdown-and-session-header.md`. + +- **`[Models]` `GET /models` is now served from cache instead of refetching on every provider poll (#222).** `ModelListFetcher.fetch()` performed a live upstream fetch on every `provideLanguageModelChatInformation` poll (VS Code polls every few hundred ms) because `MODEL_LIST_CACHE_TTL_MS` was only consulted on the failure path. The fresh snapshot is now checked before the network call; a new `invalidate()` hooks into `Refresh Models` so manual refreshes still hit the gateway. The `Models registered` log line is now emitted only when its signature changes, ending the identical-line spam. Documented in `docs/issues/94-20260908-issue222-uncached-models-poll.md`. + +- **`[Provider]` One shared output channel instead of one per request (#220).** The god-file split left a `createOutputChannel("OpenCode")` inside `prepareChatRequest()`, which runs per request — `createOutputChannel` registers a new Output-tab entry every call, so long sessions accumulated dozens of duplicate channels. `chatPrep` no longer creates channels; transports receive the provider's lazy singleton (disposed via `context.subscriptions`). Documented in `docs/issues/95-20260908-issue220-duplicate-output-channels.md`. + +- **`[Responses]` function_call / function_call_output items are now strictly paired (#216).** The gateway rejects the whole request with `No tool output found for function call ` when a `function_call` lost its output (history trim, lost `tool_call_id`) — and the converter even fabricated a `tool-` call_id that could never match. Fabricated ids are gone, and `pairResponsesFunctionCallItems()` drops orphaned calls/outputs before sending. 5 regression tests. Documented in `docs/issues/96-20260908-issue216-no-tool-output-for-function-call.md`. + +- **`[Responses]` Nested event payloads parse; zero-part failures now carry their own diagnostics (#217).** gpt-5.6-luna streams could end with zero extractable parts while the gateway billed completion tokens, triggering Copilot Chat's empty-response loop guard (verified: neither error string exists in our 0.7.4 VSIX). `response.output_text.delta` and reasoning events now unwrap nested payload objects (`delta.text`, `delta.content`, `text.value`, …), and after retries are exhausted a dedicated error names the token count, event stats, and the `[diag-sse-event-*]` path. The `[diag-empty-response]` dump no longer false-positives on healthy tool-call-only turns (tool calls flush in the transport's `finally`, after the diagnostic runs) — it now only fires when no healthy `finish_reason` was extracted. Documented in `docs/issues/97-20260908-issue217-luna-zero-parts-empty-response-loop.md`. + +- **`[Retry]` 429 responses honoring `Retry-After` are retried once transparently (#221).** Upstream Console Go rate limits are provider-side, but when the upstream names a short wait the extension now waits (capped at 30 s via `RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS`) and retries before anything is streamed — no duplicated content, no hard failure. Longer waits keep the existing actionable error. Documented in `docs/issues/98-20260908-issue221-429-retry-after.md`. + +### Triage + +- **#218 / #223 closed as triaged.** #218: VS Code's `chat.utilityModel` / `inlineChat.defaultModel` dropdowns only list natively integrated providers; BYOK vendors can't opt in — users should use **OpenCode: Configure Utility Models** (doc 32). #223: the reported `MissingSessionID` stack trace belongs to `vizards.deepseek-v4-for-copilot`, not this extension; we have always sent `x-opencode-session`. Full analysis in `docs/issues/99-20260908-issue218-223-triage-dropdown-and-session-header.md`. + ## [0.7.4] — 2026-09-03 ### Fixed diff --git a/docs/devlog.md b/docs/devlog.md index cba03147d..4116c830a 100644 --- a/docs/devlog.md +++ b/docs/devlog.md @@ -1,6 +1,45 @@ # 🧠 OPENCODE COPILOT CHAT DEVLOG -**Branch:** `main` | **Updated:** 2026-09-03 Asia/Jakarta | **Current Phase:** v0.7.4 batch — 6 community issue fixes (#204/#206/#207/#208/#213/#214) on branch `fix/issues-204-214-batch`, pending merge. Open: PR #161 (restore API key command). +**Branch:** `fix/issues-216-223-batch` | **Updated:** 2026-09-08 Asia/Jakarta | **Current Phase:** v0.7.5 batch verified — 5 fixes + 2 triages (#216/#217/#220/#221/#222 + #218/#223), manual + automated verification passed, pending PR. + +--- + +## ✅ Manual + automated verification (2026-09-08) + +User ran the manual checklist against a live build; results all green: + +| Check | Result | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------ | +| #222 A/B/C — no log spam, zero `GET /models` during session, Refresh Models still fetches + filters | ✅ | +| #220 — after 8 requests the Output dropdown still shows exactly one `OpenCode` channel | ✅ | +| #216/#217 — 9 chained luna tool-call turns (multi-parallel `read_file`/`runSubagent`), all 200, no `No tool output found`, no empty-response loop, usage/cache recorded | ✅ | +| Regression — per-model thinking resolves (`thinkingSource=modelConfiguration`), Go usage recording + status bar update (`entries=1..9`) | ✅ | + +Follow-up commits after verification: + +- `ac7a9a1` — suppress `[diag-empty-response]` false positive on healthy tool-call-only turns (tool calls flush after the diagnostic runs; now gated on a healthy `finish_reason`). +- `b99f449` / `25f0301` — regression suites automating the checklist: history-trim × pairing invariants, the real luna event shapes (flat + nested `output_text.delta`, tool-call sequence), and header-capture tests pinning `x-opencode-session` on `GET /models`, inline completions, and `GET /usage`. +- `418c4c4` — after the gateway enforcement tip (docs/go updated 2026-09-07): all four auxiliary fetch sites now send `x-opencode-session` (persisted per-installation id); chat path keeps the per-conversation id. + +Suite: 462/462 unit tests, full lint gate pass. + +--- + +## ✅ Batch v0.7.5 — Seven Issues (#216 #217 #218 #220 #221 #222 #223) — 2026-09-08 + +**Scope:** one branch (`fix/issues-216-223-batch`), atomic commits in triage-priority order, `closes #N` in each message. Deep-dive first: issue bodies, docs/issues 88–93, git history, published 0.7.4 VSIX extraction, and local session history. + +| Issue | Root cause (verified in code/artifact) | Fix | Commit | +| ----- | ---------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------ | --------- | +| #222 | `MODEL_LIST_CACHE_TTL_MS` only consulted on failure path → live `GET /models` every UI poll | Cache-first `fetch()`, `invalidate()` for Refresh Models, log-on-change registration | `9f0eba9` | +| #220 | `createOutputChannel("OpenCode")` inside per-request `prepareChatRequest()` (refactor fallout) | Shared lazy-singleton channel from provider; field removed from chatPrep | `5d2b525` | +| #216 | Fabricated `tool-` call_id + unpaired function_call/output items after trim → gateway 400 | No fabricated ids; `pairResponsesFunctionCallItems()` 1:1 pairing pass + 5 tests | `ed53ecb` | +| #217 | Nested Responses event payloads unrecognized → zero parts, Copilot loop guard (VSIX-verified) | Nested delta unwrapping in routing normalizer + dedicated zero-part error | `573f757` | +| #221 | Upstream Console Go 429 hardened into failure despite Retry-After | `parseRetryAfterMs` + single transparent retry (≤30 s, pre-stream) | `8d8e9f0` | +| #218 | VS Code default-model dropdowns whitelist native providers only (doc 32) | Triage doc; Configure Utility Models remains the user path | docs | +| #223 | Stack trace belongs to `vizards.deepseek-v4-for-copilot`, not this extension | Triage doc; our requests always carry `x-opencode-session` | docs | + +Docs: `docs/issues/94`–`99`, CHANGELOG 0.7.5, version bump 0.7.4 → 0.7.5. --- diff --git a/docs/issues/94-20260908-issue222-uncached-models-poll.md b/docs/issues/94-20260908-issue222-uncached-models-poll.md new file mode 100644 index 000000000..a70ab0f50 --- /dev/null +++ b/docs/issues/94-20260908-issue222-uncached-models-poll.md @@ -0,0 +1,57 @@ +# Issue #222 — Uncached `GET /models` on every `provideLanguageModelChatInformation` poll + +**Status:** ✅ Solved (branch `fix/issues-216-223-batch`, commit `9f0eba9`) +**Topic:** models / caching / provider-poll +**Updated:** 2026-09-08 +**Tags:** #models #cache #polling #logs +**GitHub Issue:** [ltmoerdani/opencode-copilot-chat#222](https://github.com/ltmoerdani/opencode-copilot-chat/issues/222) +**Related:** issue doc [35 (model list fetch resilience / #78)](35-20260720-issue78-model-list-fetch-resilience.md) + +--- + +## Problem + +Every UI refresh triggered a real upstream `GET /models` fetch plus a repeated +`Models registered` log line. The Output channel became noise; upstream saw +unbounded request volume from poll-happy clients. + +## Analysis + +`ModelListFetcher.fetch()` (`src/provider/modelList.ts`) performed the live +fetch unconditionally. `MODEL_LIST_CACHE_TTL_MS` existed but was only consulted +on the **failure** path (`loadCached()` inside `fallback()` and after the retry +loop). A successful fetch cached the snapshot, but the next `fetch()` never +looked at it. VS Code polls `provideLanguageModelChatInformation` every few +hundred ms, so each poll = live fetch + `replaceLiveModelMetadata` rebuild + +re-registration. + +## Fix + +- Cache-first reorder: `fetch()` consults the fresh snapshot + (in-memory, then `globalState`, both TTL-guarded) **before** the network + call; stale snapshots fall through to the existing fetch + retry path. +- `ModelListFetcher.invalidate()` clears both cache layers; + `refreshMetadataAndModels()` (the `Refresh Models` command) calls it so a + manual refresh still performs a real upstream fetch. +- `Models registered` summary in `src/provider/modelInfo.ts` is now logged + only when its signature (count/first/last/variant) changes. + +## Files Changed + +| File | Change | +| ---------------------------------- | --------------------------------------------------------- | +| `src/provider/modelList.ts` | cache-first `fetch()`, new `invalidate()` | +| `src/provider/modelInfo.ts` | log-on-change via `LAST_REGISTRATION_LOG_SIGNATURE` | +| `src/provider/OpenCodeProvider.ts` | `refreshMetadataAndModels()` calls `fetcher.invalidate()` | + +## Verification + +- `npx tsc --noEmit` clean; 454/454 unit tests pass; staged-lint gate pass. +- Follow-up (manual): confirm with the Output channel that repeated picker + refreshes produce no new `Models registered` lines and no `/models` traffic + within the TTL window. + +## Lessons Learned + +A TTL constant that only guards the failure path is not a cache — the +cache check has to sit before the work it is supposed to prevent. diff --git a/docs/issues/95-20260908-issue220-duplicate-output-channels.md b/docs/issues/95-20260908-issue220-duplicate-output-channels.md new file mode 100644 index 000000000..b494de03d --- /dev/null +++ b/docs/issues/95-20260908-issue220-duplicate-output-channels.md @@ -0,0 +1,51 @@ +# Issue #220 — Dozens of duplicate "OpenCode" output channels accumulate per session + +**Status:** ✅ Solved (branch `fix/issues-216-223-batch`, commit `5d2b525`) +**Topic:** output-channel / lifecycle +**Updated:** 2026-09-08 +**Tags:** #output-channel #regression #refactor-fallout +**GitHub Issue:** [ltmoerdani/opencode-copilot-chat#220](https://github.com/ltmoerdani/opencode-copilot-chat/issues/220) +**Related:** PR #155 god-file split (doc 67), PR #138 central config refactor (doc 66) + +--- + +## Problem + +Every chat request registered a brand-new `OpenCode` entry in the Output-tab +dropdown. Long sessions accumulated dozens of channels and the machine slowed +down; disabling the extension stopped the growth (reporter's A/B test). + +## Analysis + +`vscode.window.createOutputChannel()` registers a NEW channel every call — +there is no dedup by name. `OpenCodeProvider.getOutputChannel()` already used +the correct lazy-singleton pattern, but the god-file split +(`03029a6` / `e8fb6b2`) copied a `createOutputChannel("OpenCode")` call into +`prepareChatRequest()` (`src/provider/chatPrep.ts`), which runs **per request**. +Each request therefore leaked one more channel. + +## Fix + +- Removed the per-request channel creation (and the `outputChannel` field) + from `chatPrep.ts`. +- `OpenCodeProvider.provideLanguageModelChatResponse()` now passes its shared + lazy-singleton channel (already pushed to `context.subscriptions`, so it is + disposed with the extension) to all four transports. + +## Files Changed + +| File | Change | +| ---------------------------------- | -------------------------------------------------- | +| `src/provider/chatPrep.ts` | drop `createOutputChannel` + `outputChannel` field | +| `src/provider/OpenCodeProvider.ts` | use `this.getOutputChannel()` for transports | + +## Verification + +- `npx tsc --noEmit` clean; 454/454 unit tests pass; staged-lint gate pass. +- Follow-up (manual): keep a session open through multiple requests and verify + exactly one `OpenCode` entry exists in the Output dropdown. + +## Lessons Learned + +`createOutputChannel` is a register operation, not a lookup — any call that +isn't behind a singleton guard will multiply UI entries. diff --git a/docs/issues/96-20260908-issue216-no-tool-output-for-function-call.md b/docs/issues/96-20260908-issue216-no-tool-output-for-function-call.md new file mode 100644 index 000000000..e406a6639 --- /dev/null +++ b/docs/issues/96-20260908-issue216-no-tool-output-for-function-call.md @@ -0,0 +1,63 @@ +# Issue #216 — "No tool output found for function call call_…" (HTTP 400, gpt-5.6-luna) + +**Status:** ✅ Solved (branch `fix/issues-216-223-batch`, commit `ed53ecb`) +**Topic:** responses-api / tool-call-pairing +**Updated:** 2026-09-08 +**Tags:** #responses #tool-calls #pairing #luna +**GitHub Issue:** [ltmoerdani/opencode-copilot-chat#216](https://github.com/ltmoerdani/opencode-copilot-chat/issues/216) +**Related:** issue doc [90 (#206 fc_ item id normalization)](90-20260903-issue206-luna-responses-fc-id-mismatch.md), issue doc [68 (history trim)](68-20260820-history-trim-context-overflow.md) + +--- + +## Problem + +The request after a tool call failed with `Upstream request failed: +[invalid_request_error] No tool output found for function call call_RITWx…` +(HTTP 400) — the whole turn died even though VS Code had delivered the tool +result. + +## Analysis + +The #206 fix (commit `40be420`) normalized the `function_call` **item id** to +the `fc_` namespace while keeping `call_id` verbatim so it can pair with +`function_call_output.call_id`. Two residual gaps broke that pairing: + +1. `responsesInputItemsFromMessage()` fabricated + `call_id: \`tool-${Date.now()}\``for tool messages whose`tool_call_id` + was lost — an id that can never match its call, so the gateway rejected + the request. +2. History replay could deliver a `function_call` without its output (or vice + versa) — e.g. after `trimOldMessagesToFitContext` dropped half a group, or + when VS Code replayed tool messages in unusual shapes. The gateway rejects + the entire request for a single unpaired item. + +## Fix + +- Never fabricate a `call_id`: a tool message without `tool_call_id` produces + no `function_call_output` item. +- New pure `pairResponsesFunctionCallItems()` enforces strict 1:1 pairing over + the assembled `input` list: orphaned outputs are dropped, calls without an + output are dropped, duplicate outputs keep only the first. Wired into + `buildResponsesRequestBody()`. +- Unit tests cover matched pairs, orphaned output, trimmed call, duplicate + output, and the missing-`tool_call_id` case (`src/test/responsesRequest.test.ts`). + +## Files Changed + +| File | Change | +| ----------------------------------- | ---------------------------------------------------------- | +| `src/responsesRequest.ts` | no fabricated call_id + `pairResponsesFunctionCallItems()` | +| `src/request/openai.ts` | pairing pass in `buildResponsesRequestBody()` | +| `src/test/responsesRequest.test.ts` | 5 regression tests | + +## Verification + +- 454/454 unit tests pass (5 new); `npx tsc --noEmit` clean; staged-lint gate pass. +- Follow-up (manual): run a tool-calling agent turn on gpt-5.6-luna via + `/v1/responses` and confirm the follow-up request no longer 400s. + +## Lessons Learned + +Fixing an id-format rejection (#206) without also enforcing the pairing +invariant left the adjacent failure mode one step away. Pairing should be a +post-condition of request assembly, not an emergent property of history. diff --git a/docs/issues/97-20260908-issue217-luna-zero-parts-empty-response-loop.md b/docs/issues/97-20260908-issue217-luna-zero-parts-empty-response-loop.md new file mode 100644 index 000000000..f16c70484 --- /dev/null +++ b/docs/issues/97-20260908-issue217-luna-zero-parts-empty-response-loop.md @@ -0,0 +1,63 @@ +# Issue #217 — gpt-5.6-luna consumes tokens but returns nothing ("empty-response loop") + +**Status:** ✅ Solved (branch `fix/issues-216-223-batch`, commit `573f757`) — ⚠️ root-cause payload confirmed indirectly; keep the `[diag-sse-event-*]` path if a new shape appears +**Topic:** responses-api / zero-parts / diagnostics +**Updated:** 2026-09-08 +**Tags:** #responses #luna #streaming #diagnostics +**GitHub Issue:** [ltmoerdani/opencode-copilot-chat#217](https://github.com/ltmoerdani/opencode-copilot-chat/issues/217) +**Related:** issue doc [86 (#197/#198 responses finish-reason + text extraction)](86-20260828-issue197-198-responses-api-finish-reason-text-extraction.md), issue doc [41 (luna routing)](41-20260803-gpt56-luna-routing-fix.md) + +--- + +## Problem + +OpenCode Go billed 146 completion tokens on gpt-5.6-luna but the chat showed +nothing, then VS Code surfaced "The request was stopped to prevent an +empty-response loop". Stack frames pointed at `engine.js` in our build. + +## Analysis + +Both strings in the reported error ("returned no usable response after +consuming N completion tokens", "prevent an empty-response loop") do **not** +exist anywhere in our source **or in the published 0.7.4 VSIX** (verified by +extracting the Marketplace package and grepping `out/`). The loop-guard wording +is Copilot Chat core reacting to a provider response that emitted zero chat +parts — i.e. our extractor produced nothing for a stream the gateway had +billed. This is the same failure class as #93 / #197 / #198: a new upstream +payload shape that the Responses event normalizer doesn't recognize. + +Audit of `normalizeResponsesStreamEvent()` found a real gap: +`response.output_text.delta` only read **string-shaped** payloads +(`data.delta`, `data.text`). A nested shape (`delta: { text | content | value }` +or `text: { value }`) returned `{choices: []}` → zero parts for the whole +stream. Reasoning extraction had the same string-only assumption. + +## Fix + +- `response.output_text.delta` now unwraps nested payload objects before + falling back to the flat fields. +- `extractResponsesReasoningText()` accepts nested `delta.text/thinking/summary`. +- After the bounded stream-failure retries are exhausted, a dedicated + zero-part error names the token count, event/byte stats, and the + `[diag-sse-event-*]` diagnostic path, so any future unhandled shape is + reportable in one round-trip instead of surfacing as a generic loop guard. + +## Files Changed + +| File | Change | +| -------------------------- | -------------------------------------------------------- | +| `src/core/routing.ts` | nested-payload support in text + reasoning normalization | +| `src/transports/engine.ts` | dedicated zero-parts-with-tokens error after retries | + +## Verification + +- `npx tsc --noEmit` clean; 454/454 unit tests pass; staged-lint gate pass. +- Follow-up (manual): a luna session that previously hit the loop guard should + now deliver content; if the gateway ships yet another shape, the new error + message points reporters at the exact diag lines needed. + +## Lessons Learned + +"Zero parts + billed tokens" is always an extraction gap, never a model +refusal — treat the normalizer's fallback `{choices: []}` as the first +suspect, and make the failure message carry its own diagnostics. diff --git a/docs/issues/98-20260908-issue221-429-retry-after.md b/docs/issues/98-20260908-issue221-429-retry-after.md new file mode 100644 index 000000000..c50891f1c --- /dev/null +++ b/docs/issues/98-20260908-issue221-429-retry-after.md @@ -0,0 +1,58 @@ +# Issue #221 — "Rate limit exceeded" on OpenCode Go even with minimal usage + +**Status:** ✅ Solved (branch `fix/issues-216-223-batch`, commit `8d8e9f0`) +**Topic:** rate-limit / retry +**Updated:** 2026-09-08 +**Tags:** #429 #retry #rate-limit #go +**GitHub Issue:** [ltmoerdani/opencode-copilot-chat#221](https://github.com/ltmoerdani/opencode-copilot-chat/issues/221) +**Related:** PR #107 transient 5xx retry (doc 51), issue doc [94 (#222 model-list caching)](94-20260908-issue222-uncached-models-poll.md) + +--- + +## Problem + +`OpenCode Go API rate/quota limit (429) … Error from provider (Console Go): +Upstream request failed: [rate_limit_exceeded] … no quota headers` appeared +with minimal usage. + +## Analysis + +The 429 originates **upstream of the gateway** (the Console Go provider model +itself), and the response carries no quota headers, so the extension cannot +know the window — this is not a bug in the extension's usage accounting. +What the extension _could_ do better: + +1. A 429 became a hard user-facing failure even when the upstream answered + with a `Retry-After` naming a short wait. +2. Issue #222 aggravated per-key request volume: every provider poll fired a + real `GET /models` (fixed in doc 94). + +## Fix + +- New `parseRetryAfterMs()` (`src/utils.ts`) accepts delta-seconds and + HTTP-date forms. +- In `streamOpenCodeResponse()`, a 429 with a `Retry-After` of at most + `RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS` (30 s, `src/config.ts`) is waited out + once and transparently retried. This happens **before any content is + streamed**, so a retry cannot duplicate chat output. Longer waits still + surface the existing actionable rate-limit error. + +## Files Changed + +| File | Change | +| -------------------------- | ------------------------------------------ | +| `src/utils.ts` | `parseRetryAfterMs()` | +| `src/config.ts` | `RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS` | +| `src/transports/engine.ts` | 429 + Retry-After single transparent retry | + +## Verification + +- `npx tsc --noEmit` clean; 454/454 unit tests pass; staged-lint gate pass. +- Follow-up (manual): observe `[rate-limit]` + `[retry] 429 … honoring +Retry-After` lines in the Output channel when the upstream throttles. + +## Lessons Learned + +When the upstream is the limiter, the client-side win is honoring its hints — +and removing self-inflicted request volume (the #222 poll fetch) matters as +much as retry policy. diff --git a/docs/issues/99-20260908-issue218-223-triage-dropdown-and-session-header.md b/docs/issues/99-20260908-issue218-223-triage-dropdown-and-session-header.md new file mode 100644 index 000000000..06ca3fa9e --- /dev/null +++ b/docs/issues/99-20260908-issue218-223-triage-dropdown-and-session-header.md @@ -0,0 +1,71 @@ +# Issues #218 / #223 — Triage: VS Code dropdown limitation & third-party extension misreport + +**Status:** ✅ Solved (triage — no extension defect; branch `fix/issues-216-223-batch`, commit with docs) +**Topic:** triage / default-model-settings / misattribution +**Updated:** 2026-09-08 +**Tags:** #triage #utility-model #byok #session-headers +**GitHub Issues:** [#218](https://github.com/ltmoerdani/opencode-copilot-chat/issues/218) · [#223](https://github.com/ltmoerdani/opencode-copilot-chat/issues/223) +**Related:** issue doc [32 (VS Code 1.128 BYOK utility model)](32-20260708-vscode-128-byok-utility-model.md), `src/request/headers.ts` + +--- + +## #218 — Models missing from `inlineChat.defaultModel` / `chat.utilityModel` / … dropdowns + +Not an extension defect. Those settings render a **closed dropdown** that only +lists models from providers VS Code itself integrates (Copilot, OpenRouter +built-in). BYOK providers registered through the `chatProvider` API — which is +how OpenCode Go/Zen models appear at all — are not offered in that picker; +this is a VS Code limitation, not something the extension can opt into +(`isUserSelectable: true` is already set on every model we register). + +`chat.defaultModel` works for the reporter because it is a free-text field. + +Resolution path for users is the existing **OpenCode: Configure Utility +Models** command (shipped after the investigation in doc 32), which sets +`chat.byokUtilityModelDefault` / `chat.utilityModel` / `chat.utilitySmallModel` +explicitly. Closing as "works as designed / upstream limitation"; if a future +VS Code API exposes dropdown registration for BYOK vendors, we can revisit. + +## #223 — "[400] Invalid request body format … Request is missing x-opencode-session" + +Misattributed to this extension. Decisive evidence in the report itself — the +stack trace belongs to a **different extension**: + +```text +at DeepSeekClient.streamChatCompletion + (/home/marco/.vscode/extensions/vizards.deepseek-v4-for-copilot-0.8.2/ + out/client/core.js:46:23) +``` + +Our extension has always sent `x-opencode-session` on every request +(`buildOpenCodeRequestHeaders()`, `src/request/headers.ts:40`), with a +stable `vscode-` fallback when VS Code does not expose a session id, +and every request line in our Output channel logs the `session=` value. The +`MissingSessionID` requirement appears to be new gateway guidance +([OpenCode Go docs — where can I use it](https://opencode.ai/docs/go/#where-can-i-use-it)) that the third-party +extension does not yet satisfy. + +Resolution: close with an explanation and point the reporter at +`OpenCode Go: Diagnostics` (which proves our requests carry the session +header) in case their report was a mix-up between two installed extensions. + +### Follow-up (2026-09-08): auxiliary requests were missing the header + +While the main chat path always sent `x-opencode-session`, an audit triggered +by the same gateway enforcement found **four auxiliary fetch sites** hitting +the gateway without it: `GET /models` (`ModelListFetcher`), inline completions +(`ChatCompletionEngine`), `GET /usage` (`fetchGoUsage`), and the +Manage-Provider test connection. The Go docs (updated 2026-09-07) now state +the requirement explicitly, and per the reporter's note requests missing the +header may error from 2026-09-06. Fixed in commit `8c138e2`: all four now send +a persisted per-installation session id (`auxiliarySessionId()`, +globalState-backed) — no conversation context exists for these requests, so a +stable installation id is the correct affinity key. + +## Lessons Learned + +1. Extension-version fields in issue templates report the _installed set_, not + which binary produced a stack trace — always read the full stack before + accepting attribution. +2. Closed dropdowns in VS Code settings are provider-whitelists, not model + queries; BYOK workarounds must route through free-text settings. diff --git a/package-lock.json b/package-lock.json index 2be41a540..c9c3baf98 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "opencode-copilot-chat", - "version": "0.7.3", + "version": "0.7.5", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "opencode-copilot-chat", - "version": "0.7.3", + "version": "0.7.5", "license": "MIT", "dependencies": { "@silvia-odwyer/photon-node": "^0.3.4" @@ -15,7 +15,7 @@ "@types/node": "^26.4.0", "@types/vscode": "^1.125.0", "@vscode/vsce": "^3.9.2", - "editorconfig-checker": "^6.1.1", + "editorconfig-checker": "^6.2.0", "eslint": "^10.8.1", "eslint-plugin-jsonc": "^3.4.1", "eslint-plugin-yml": "^3.8.1", @@ -3058,9 +3058,9 @@ } }, "node_modules/editorconfig-checker": { - "version": "6.1.1", - "resolved": "https://registry.npmjs.org/editorconfig-checker/-/editorconfig-checker-6.1.1.tgz", - "integrity": "sha512-kiOb6qaWpMNt7Z/43ba0Pa1Inhr2/t9nKbvEKtCeXJ5AesztoM9AgLOOQVB4QUv/nGjgz3xkbx4pcogVRD2NWw==", + "version": "6.2.0", + "resolved": "https://registry.npmjs.org/editorconfig-checker/-/editorconfig-checker-6.2.0.tgz", + "integrity": "sha512-5zrNwlxUWyOvAcrSK4mPIGNYrf1KK5pH2D3VXDB1AcDsDdBfLNPOjBL5jHgYh+JykOIuXlQHMeYvXPrZwGNmHA==", "dev": true, "license": "MIT", "bin": { diff --git a/package.json b/package.json index 13e4f5879..da5446788 100644 --- a/package.json +++ b/package.json @@ -2,7 +2,7 @@ "name": "opencode-copilot-chat", "displayName": "OpenCode for Copilot Chat — BYOK 30+ AI Models", "description": "Use 30+ frontier AI models (DeepSeek V4, Kimi K2.6, GLM-5.1, Qwen3.7, MiMo V2.5, MiniMax M2.7, free Claude Opus, GPT-5.5, Gemini 3.5, Grok) in GitHub Copilot Chat. Bring Your Own Key — no Copilot Pro needed.", - "version": "0.7.4", + "version": "0.7.5", "publisher": "ltmoerdani", "license": "MIT", "icon": "media/opencodego.png", @@ -545,7 +545,7 @@ "@types/node": "^26.4.0", "@types/vscode": "^1.125.0", "@vscode/vsce": "^3.9.2", - "editorconfig-checker": "^6.1.1", + "editorconfig-checker": "^6.2.0", "eslint": "^10.8.1", "eslint-plugin-jsonc": "^3.4.1", "eslint-plugin-yml": "^3.8.1", diff --git a/src/autocomplete/engine.ts b/src/autocomplete/engine.ts index 53efd862f..d47f28369 100644 --- a/src/autocomplete/engine.ts +++ b/src/autocomplete/engine.ts @@ -17,6 +17,8 @@ export interface ChatCompletionEngineOptions { /** Gateway chat-completions URL (provider-specific). */ chatCompletionsUrl: string; apiKey: string; + /** Stable session id for the gateway's x-opencode-session enforcement. */ + sessionId?: string; timeoutMs?: number; log?: (msg: string) => void; } @@ -77,6 +79,9 @@ export class ChatCompletionEngine implements CompletionEngine { headers: { Authorization: `Bearer ${this.options.apiKey}`, "Content-Type": "application/json", + // Gateway enforcement (docs/go): all OpenCode requests need a + // session id; completions share the persisted per-installation id. + ...(this.options.sessionId ? { "x-opencode-session": this.options.sessionId } : {}), }, body: JSON.stringify(body), signal: AbortSignal.any([signal, AbortSignal.timeout(this.timeoutMs)]), diff --git a/src/autocomplete/index.ts b/src/autocomplete/index.ts index 01a93ce6b..4b3bcdf52 100644 --- a/src/autocomplete/index.ts +++ b/src/autocomplete/index.ts @@ -32,6 +32,7 @@ import { SETTING_INLINE_SUGGESTIONS_CHAT_INPUT, } from "../config"; import { toFiniteNumber } from "../utils"; +import { auxiliarySessionId } from "../request/headers"; import { bumpCompletionUsage, matchesAcceptance, utcDayStart, type CompletionUsageDay } from "./usage"; export { @@ -121,6 +122,7 @@ export function registerInlineCompletions(context: vscode.ExtensionContext, deps const keyed = new ChatCompletionEngine({ chatCompletionsUrl: deps.chatCompletionsUrl, apiKey, + sessionId: auxiliarySessionId(context), timeoutMs: readNumberSetting(INLINE_TIMEOUT_MS_SETTING, DEFAULT_INLINE_TIMEOUT_MS, 500, 15_000), log: (msg) => { log(msg); diff --git a/src/config.ts b/src/config.ts index 672415afe..5d63abbd3 100644 --- a/src/config.ts +++ b/src/config.ts @@ -302,6 +302,9 @@ export const TRANSIENT_5XX_MAX_RETRIES = 2; export const TRANSIENT_5XX_RETRY_BASE_MS = 1000; export const TRANSIENT_5XX_RETRY_JITTER_MS = 250; +/** Maximum wait the extension will honor from a 429 Retry-After header (issue #221). */ +export const RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS = 30_000; + // ─── Transient network (fetch) retry for chat requests (engine.ts) ─────────── // Mirrors the model-list fetch resilience (issue #78): a `fetch()` that *throws* // (undici `TypeError: fetch failed` — ECONNRESET / EAI_AGAIN / UND_ERR_CONNECT_TIMEOUT diff --git a/src/core/routing.ts b/src/core/routing.ts index fc4e8948c..f524ae5cd 100644 --- a/src/core/routing.ts +++ b/src/core/routing.ts @@ -56,7 +56,12 @@ export function normalizeResponsesStreamEvent(data: unknown): unknown { } if (eventType === "response.output_text.delta") { - const delta = firstStringRaw(data.delta, data.text, data.output_text_delta); + // Some gateways nest the payload (delta: { text } or text: { value }) + // instead of sending a flat string; handle both so models like Luna + // don't come through as a zero-part stream (issue #217). + const nestedDelta = isRecord(data.delta) ? firstStringRaw(data.delta.text, data.delta.content, data.delta.value) : undefined; + const nestedText = isRecord(data.text) ? firstStringRaw(data.text.value, data.text.content) : undefined; + const delta = firstStringRaw(data.delta, nestedDelta, data.text, nestedText, data.output_text_delta); return delta ? { choices: [ @@ -393,7 +398,10 @@ function normalizeResponsesUsage(usage: unknown): Record | unde * until the gateway relays plaintext reasoning. See `src/thinking/muse.ts`. */ function extractResponsesReasoningText(data: Record): string { - const direct = firstStringRaw(data.delta, data.text, data.summary_text, data.output_text_delta); + // Nested shapes (delta: { text }) appear alongside the flat ones on newer + // gateways — accept both (issue #217). + const nestedDelta = isRecord(data.delta) ? firstStringRaw(data.delta.text, data.delta.thinking, data.delta.summary) : undefined; + const direct = firstStringRaw(data.delta, nestedDelta, data.text, data.summary_text, data.output_text_delta); if (direct) { return direct; } diff --git a/src/provider/OpenCodeProvider.ts b/src/provider/OpenCodeProvider.ts index 018d78731..0db3745f3 100644 --- a/src/provider/OpenCodeProvider.ts +++ b/src/provider/OpenCodeProvider.ts @@ -148,6 +148,9 @@ export class OpenCodeProvider implements vscode.LanguageModelChatProvider { await clearOpenCodeModelMetadataCache(this.context); + // Bypass the cache-first short-circuit so "Refresh Models" always + // performs a real upstream fetch (issue #222). + this.fetcher.invalidate(); // Pass the stored API key so the gateway sees the authenticated // (per-key) model list, not the anonymous default. const apiKey = await this.context.secrets.get(secretKeyFor(this.baseVendor)); @@ -324,10 +327,13 @@ export class OpenCodeProvider implements vscode.LanguageModelChatProvider; thinkingPayload: unknown; requestHeaders: Record; - outputChannel: vscode.OutputChannel; onTransportSummary: (summary: TransportRequestSummary) => void; }> { // VS Code can invoke a cached selected model immediately after the @@ -277,7 +276,11 @@ export async function prepareChatRequest( endpoint: routing.endpointKind === "messages" ? "messages" : routing.endpointKind === "responses" ? "responses" : "chat", }); const requestHeaders = buildOpenCodeRequestHeaders(messages, options, rawModelId); - const outputChannel = vscode.window.createOutputChannel("OpenCode"); + // NOTE: no output channel is created here. Channels are process-wide singletons + // (vscode.window.createOutputChannel registers a new Output-tab entry every + // call), so creating one per request flooded the Output tab with dozens of + // duplicate "OpenCode" channels (issue #220). Transports receive the + // provider's shared channel instead. const onTransportSummary = (summary: TransportRequestSummary) => { // Compute credits for VS Code session cost (1 credit = $0.01). // VS Code reads usage.copilotCredits from the LanguageModelDataPart @@ -326,7 +329,6 @@ export async function prepareChatRequest( limits, thinkingPayload, requestHeaders, - outputChannel, onTransportSummary, }; } diff --git a/src/provider/modelInfo.ts b/src/provider/modelInfo.ts index 82c3e4f22..87811ecca 100644 --- a/src/provider/modelInfo.ts +++ b/src/provider/modelInfo.ts @@ -23,6 +23,12 @@ import { } from "./settings"; import { ensureProfileSync } from "../usage/dashboard"; +// ISSUE #222: VS Code calls provideLanguageModelChatInformation on every UI +// poll. Log the "Models registered" summary only when it actually changes +// (model count, first/last id, variant) so the Output channel stays a +// signal source instead of repeating identical lines. +const LAST_REGISTRATION_LOG_SIGNATURE = new Map(); + /** * Body of `provideLanguageModelChatInformation`: BYOK/secret key resolution, * profile registration and per-model `OpenCodeModel` assembly. Pure with @@ -228,11 +234,15 @@ export async function provideModelChatInformation( // model ID so we can still debug registration issues without flooding // the Output channel when VS Code refreshes model info frequently. if (registeredCount > 0) { - deps.log( - `Models registered: count=${String(registeredCount)} provider=${deps.definition.vendor}` + - ` first=${firstModelId} last=${lastModelId}` + - (deps.definition.isAgentVariant ? " (agents)" : ""), - ); + const signature = + `count=${String(registeredCount)} provider=${deps.definition.vendor}` + + ` first=${firstModelId} last=${lastModelId}` + + (deps.definition.isAgentVariant ? " (agents)" : ""); + const logKey = deps.definition.isAgentVariant ? `${deps.definition.vendor}::agents` : deps.definition.vendor; + if (LAST_REGISTRATION_LOG_SIGNATURE.get(logKey) !== signature) { + LAST_REGISTRATION_LOG_SIGNATURE.set(logKey, signature); + deps.log(`Models registered: ${signature}`); + } } return results; diff --git a/src/provider/modelList.ts b/src/provider/modelList.ts index 60436ecc7..7b115b2a5 100644 --- a/src/provider/modelList.ts +++ b/src/provider/modelList.ts @@ -9,6 +9,7 @@ import { import { getErrorMessage } from "../utils"; import { sleep } from "../utils"; import { getUserAgent, isTransientFetchError, type ModelListEntry, type ModelListResponse, type ProviderDefinition } from "./definitions"; +import { auxiliarySessionId } from "../request/headers"; import { resolveBaseVendor } from "../providerTypes"; /** @@ -35,6 +36,17 @@ export class ModelListFetcher { async fetch(apiKey?: string, token?: vscode.CancellationToken): Promise { if (token?.isCancellationRequested) return this.fallback(); + // ISSUE #222: VS Code polls provideLanguageModelChatInformation every + // few hundred ms. Consult the fresh cached snapshot BEFORE the live + // fetch so each poll is local work instead of an upstream + // `GET /models` round-trip. A stale snapshot falls through to the + // live fetch + retry path below; `Refresh Models` forces a real + // refresh by clearing the cache first. + const cachedFresh = this.loadCached(); + if (cachedFresh) { + return this.deps.filterAvailableModels(cachedFresh.ids); + } + // Explicit Accept + User-Agent make this look like a legitimate API call // rather than an anonymous scanner. Some corporate firewalls / SSL // inspection proxies (Zscaler, Netskope, Fortinet) drop bare GETs that @@ -43,6 +55,10 @@ export class ModelListFetcher { const headers: Record = { "User-Agent": getUserAgent(), Accept: "application/json", + // Gateway enforcement: every OpenCode request needs a stable session id + // (docs/go, 2026-09-07). /models has no conversation — use the + // persisted per-installation id. + "x-opencode-session": auxiliarySessionId(this.deps.context), }; if (apiKey) { headers["Authorization"] = `Bearer ${apiKey}`; @@ -166,6 +182,17 @@ export class ModelListFetcher { return this.deps.filterAvailableModels(this.deps.definition.fallbackModels); } + /** + * Drop the cached snapshot (in-memory + globalState) so the next + * {@link fetch} performs a real upstream request. Used by the manual + * `Refresh Models` command, which must bypass the cache-first + * short-circuit added for issue #222. + */ + invalidate(): void { + this.cached = undefined; + void this.deps.context.globalState.update(this.cacheKey, undefined); + } + /** * Read the last successful fetch from in-memory cache or globalState. * Returns undefined when absent or past {@link MODEL_LIST_CACHE_TTL_MS}. diff --git a/src/provider/providerDialogs.ts b/src/provider/providerDialogs.ts index a66a0363f..4cc8d212f 100644 --- a/src/provider/providerDialogs.ts +++ b/src/provider/providerDialogs.ts @@ -1,6 +1,7 @@ import * as vscode from "vscode"; import { TEST_CONNECTION_TIMEOUT_MS, secretKeyFor } from "../config"; import { getErrorMessage } from "../utils"; +import { auxiliarySessionId } from "../request/headers"; import type { ProviderVendor } from "../providerTypes"; import type { ProviderDefinition } from "./definitions"; import { configureUtilityModels, toggleProviderEnabled } from "../commands/providers"; @@ -86,6 +87,8 @@ export async function testConnection(deps: DialogDeps): Promise { headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json", + // Gateway enforcement (docs/go): auxiliary requests need a session id. + "x-opencode-session": auxiliarySessionId(deps.context), }, body: JSON.stringify({ model: deps.definition.testModelId, diff --git a/src/request/headers.ts b/src/request/headers.ts index dc0bec192..6135a623e 100644 --- a/src/request/headers.ts +++ b/src/request/headers.ts @@ -1,9 +1,35 @@ import * as vscode from "vscode"; +import { randomUUID } from "node:crypto"; import { OPEN_CODE_CLIENT } from "../config"; import { getUserAgent } from "../provider/definitions"; import { messageText } from "../provider/tokens"; import { isRecord } from "../utils"; +// OpenCode Go gateway enforcement (docs/go, updated 2026-09-07): every request +// to the gateway should carry a stable `x-opencode-session`. The main chat +// path builds a per-conversation id in buildOpenCodeRequestHeaders(); the +// auxiliary requests below (/models, /usage, test-connection, inline +// completions) have no conversation, so they share one persisted +// per-installation id instead. + +/** globalState key for the persisted auxiliary session id. */ +const AUX_SESSION_STATE_KEY = "opencode.auxSessionId"; + +/** + * Stable per-installation session id for auxiliary gateway requests that have + * no conversation context (/models, /usage, test connection, inline + * completions). Generated once and persisted in globalState. + */ +export function auxiliarySessionId(context: vscode.ExtensionContext): string { + const existing = context.globalState.get(AUX_SESSION_STATE_KEY); + if (existing && existing.trim()) { + return cleanHeaderValue(existing); + } + const id = cleanHeaderValue(`vscode-aux-${randomUUID()}`); + void context.globalState.update(AUX_SESSION_STATE_KEY, id); + return id; +} + // The official OpenCode client sends these headers on every request. The Zen // gateway reads x-opencode-session first, then converts that sticky identifier // into provider-specific affinity headers such as x-session-affinity upstream. diff --git a/src/request/openai.ts b/src/request/openai.ts index b89957552..964cfdc8f 100644 --- a/src/request/openai.ts +++ b/src/request/openai.ts @@ -9,7 +9,7 @@ */ import * as vscode from "vscode"; import { lookupModelRegistryEntry } from "../core/registry"; -import { buildResponsesRequestEnvelope, responsesInputItemsFromMessage } from "../responsesRequest"; +import { buildResponsesRequestEnvelope, pairResponsesFunctionCallItems, responsesInputItemsFromMessage } from "../responsesRequest"; import { thinkingProviderFor } from "../thinking"; import { sanitizeToolSchema } from "./schema"; import { messagesHaveImages } from "./shared"; @@ -52,7 +52,10 @@ export function buildResponsesRequestBody( metadata: ResolvedModelMetadata, limits: ModelLimits, ): Record { - const input = messages.flatMap((message) => responsesInputItemsFromMessage(message)); + // ISSUE #216: drop function_call / function_call_output items that lost + // their partner (history trim, lost tool_call_id) — the upstream rejects + // the whole request otherwise. + const input = pairResponsesFunctionCallItems(messages.flatMap((message) => responsesInputItemsFromMessage(message))); const tools = mapResponsesTools(options.tools, modelId); const thinkingPayload = thinkingProviderFor(modelId).buildPayload(settings.thinking, { hasImageInput: messagesHaveImages(messages), diff --git a/src/responsesRequest.ts b/src/responsesRequest.ts index dc2ea22d1..25647fdef 100644 --- a/src/responsesRequest.ts +++ b/src/responsesRequest.ts @@ -105,10 +105,15 @@ export function responsesInputItemsFromMessage(message: ResponsesApiMessage): Re // Vision-capable OpenAI/Anthropic/Google transports handle images in tool // results natively via their respective request builders. const output = typeof message.content === "string" ? message.content : responsesToolOutput(message.content); + // RULES (issue #216): never fabricate a call_id. A random placeholder can + // never match the paired `function_call.call_id`, and the gateway rejects + // the entire request with `No tool output found for function call`. + // Unpaired outputs are dropped by {@link pairResponsesFunctionCallItems}. + if (!message.tool_call_id) return []; return [ { type: "function_call_output", - call_id: message.tool_call_id ?? `tool-${String(Date.now())}`, + call_id: message.tool_call_id, output, }, ]; @@ -117,6 +122,52 @@ export function responsesInputItemsFromMessage(message: ResponsesApiMessage): Re return []; } +/** + * Enforce 1:1 pairing between `function_call` and `function_call_output` + * items (issue #216). The Console Go upstream rejects the WHOLE request with + * `No tool output found for function call ` when a `function_call` has no + * matching output — e.g. after history trimming dropped the tool result, or + * when VS Code replays a tool message whose `tool_call_id` was lost. + * + * RULES: + * - A `function_call_output` whose `call_id` matches no `function_call` is + * dropped (stale/orphaned result). + * - A `function_call` with no output is dropped too (the gateway 400s on it; + * keeping it would fail the entire turn). + * - The first output wins if a call_id is duplicated. + */ +export function pairResponsesFunctionCallItems(items: Record[]): Record[] { + const callIdsWithOutput = new Set(); + const seenOutputs = new Set(); + for (const item of items) { + if (item.type === "function_call_output" && typeof item.call_id === "string") { + if (!seenOutputs.has(item.call_id)) { + seenOutputs.add(item.call_id); + callIdsWithOutput.add(item.call_id); + } + } + } + const callIdsWithCall = new Set(); + for (const item of items) { + if (item.type === "function_call" && typeof item.call_id === "string") { + callIdsWithCall.add(item.call_id); + } + } + const consumedOutputs = new Set(); + return items.filter((item) => { + const callId = item.call_id; + if (item.type === "function_call_output") { + const matched = typeof callId === "string" && callIdsWithCall.has(callId) && !consumedOutputs.has(callId); + if (matched) consumedOutputs.add(callId); + return matched; + } + if (item.type === "function_call") { + return typeof callId === "string" && callIdsWithOutput.has(callId); + } + return true; + }); +} + /** * Return a Responses-API-compliant `function_call` item id. Ids already in the * `fc_` namespace pass through unchanged; anything else (chat-completions diff --git a/src/test/helpers/goUsageTestUtils.ts b/src/test/helpers/goUsageTestUtils.ts index 66a62ab9c..544181d62 100644 --- a/src/test/helpers/goUsageTestUtils.ts +++ b/src/test/helpers/goUsageTestUtils.ts @@ -96,6 +96,9 @@ export function installVscodeMock(): void { module.exports = { ExtensionContext: class {}, MarkdownString, + // getUserAgent() probes the extension's package.json for the version; + // returning undefined falls back to the default UA string. + extensions: { getExtension: () => undefined }, }; `, "utf-8", diff --git a/src/test/issue216-217-regression.test.ts b/src/test/issue216-217-regression.test.ts new file mode 100644 index 000000000..ca46f2b57 --- /dev/null +++ b/src/test/issue216-217-regression.test.ts @@ -0,0 +1,225 @@ +import { describe, it, before } from "node:test"; +import assert from "node:assert/strict"; +import Module from "node:module"; +import path from "node:path"; +import fs from "node:fs"; +import os from "node:os"; + +/* + * Regression tests for the issue batch #216 / #217, based on the REAL + * gpt-5.6-luna event shapes captured in the Output channel during manual + * verification (2026-09-08, /v1/responses tool-calling sessions). + * + * - #216 Test B: a long session that triggers history trim must still produce + * strictly paired function_call / function_call_output items (the gateway + * rejects the whole request otherwise). + * - #217: the real luna event shapes must extract into usable parts (tool + * calls / text), including flat AND nested output_text.delta payloads. + */ + +const vscodeMockPath = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "vscode-mock-216-")), "index.js"); +fs.writeFileSync( + vscodeMockPath, + `"use strict"; +class LanguageModelTextPart { constructor(value) { this.value = value; } } +class LanguageModelThinkingPart { constructor(text) { this.text = text; } } +class LanguageModelToolCallPart { + constructor(callId, name, input) { this.callId = callId; this.name = name; this.input = input; } +} +class LanguageModelToolResultPart { constructor(callId, content) { this.callId = callId; this.content = content; } } +module.exports = { LanguageModelTextPart, LanguageModelThinkingPart, LanguageModelToolCallPart, LanguageModelToolResultPart }; +`, + "utf-8", +); + +type ResolveFilename = (request: string, parent: unknown, ...args: unknown[]) => string; +const moduleResolver = Module as unknown as { _resolveFilename: ResolveFilename }; +const originalResolveFilename = moduleResolver._resolveFilename; +moduleResolver._resolveFilename = function (request: string, parent: unknown, ...args: unknown[]): string { + if (request === "vscode") { + return vscodeMockPath; + } + return originalResolveFilename.call(this, request, parent, ...args); +}; + +let OpenAiResponseExtractor: typeof import("../transports/extractors.js").OpenAiResponseExtractor; +let normalizeResponsesStreamEvent: typeof import("../core/routing.js").normalizeResponsesStreamEvent; +let trimOldMessagesToFitContext: typeof import("../provider/historyTrim.js").trimOldMessagesToFitContext; +let historyByteCapForBudget: typeof import("../provider/historyTrim.js").historyByteCapForBudget; +let pairResponsesFunctionCallItems: typeof import("../responsesRequest.js").pairResponsesFunctionCallItems; +let responsesInputItemsFromMessage: typeof import("../responsesRequest.js").responsesInputItemsFromMessage; + +type ApiMessage = Record; + +/** Build a tool-call group (assistant tool_calls + tool result) as ApiMessages. */ +function toolCallGroup(callId: string, fileName: string, resultChars: number): ApiMessage[] { + return [ + { + role: "assistant", + content: null, + tool_calls: [ + { + id: callId, + type: "function", + function: { name: "read_file", arguments: JSON.stringify({ filePath: fileName, startLine: 1, endLine: 240 }) }, + }, + ], + }, + { role: "tool", tool_call_id: callId, content: "x".repeat(resultChars) }, + ]; +} + +/** Real luna event sequence for one tool call (trimmed to the discriminating shapes). */ +function lunaToolCallEvents(callId: string, argsChunks: string[]): Array> { + const events: Array> = [ + { type: "response.created", sequence_number: 0, response: { id: "resp_x", status: "in_progress", output: [], usage: null } }, + { + type: "response.output_item.added", + sequence_number: 2, + output_index: 0, + item: { id: "rs_1", type: "reasoning", encrypted_content: "gAAAA" }, + }, + { + type: "response.output_item.done", + sequence_number: 3, + output_index: 0, + item: { id: "rs_1", type: "reasoning", encrypted_content: "gAAAA" }, + }, + { + type: "response.output_item.added", + sequence_number: 4, + output_index: 1, + item: { id: "fc_1", type: "function_call", status: "in_progress", name: "read_file", call_id: callId, arguments: "" }, + }, + ]; + let seq = 5; + for (const chunk of argsChunks) { + events.push({ type: "response.function_call_arguments.delta", sequence_number: seq++, output_index: 1, item_id: "fc_1", delta: chunk }); + } + events.push({ + type: "response.function_call_arguments.done", + sequence_number: seq++, + output_index: 1, + item_id: "fc_1", + arguments: argsChunks.join(""), + }); + events.push({ + type: "response.output_item.done", + sequence_number: seq++, + output_index: 1, + item: { id: "fc_1", type: "function_call", status: "completed", name: "read_file", call_id: callId, arguments: argsChunks.join("") }, + }); + events.push({ + type: "response.completed", + sequence_number: seq++, + response: { id: "resp_x", status: "completed", stop_reason: "tool_calls", usage: { input_tokens: 100, output_tokens: 50 } }, + }); + return events; +} + +describe("#216 Test B: history trim keeps function_call pairing intact", () => { + before(async () => { + const responses = await import("../responsesRequest.js"); + pairResponsesFunctionCallItems = responses.pairResponsesFunctionCallItems; + responsesInputItemsFromMessage = responses.responsesInputItemsFromMessage; + const historyTrim = await import("../provider/historyTrim.js"); + trimOldMessagesToFitContext = historyTrim.trimOldMessagesToFitContext; + historyByteCapForBudget = historyTrim.historyByteCapForBudget; + }); + + it("never leaves an unpaired function_call/function_call_output after trimming", () => { + const messages: ApiMessage[] = [ + { role: "system", content: "You are a coding agent." }, + { role: "user", content: "Analyze the repository end to end." }, + ]; + for (let i = 0; i < 40; i++) { + messages.push(...toolCallGroup(`call_${String(i).padStart(2, "0")}`, `/repo/file-${String(i)}.ts`, 4000)); + messages.push({ role: "user", content: `Continue the analysis (step ${String(i)}). ${"y".repeat(500)}` }); + } + messages.push({ role: "user", content: "Final answer please." }); + + const budget = 2000; + // trimOldMessagesToFitContext mutates the array in place. + const trimmed = trimOldMessagesToFitContext(messages as never, budget, historyByteCapForBudget(budget)); + assert.ok(trimmed.removed > 0, "expected history trim to remove messages"); + + const items = pairResponsesFunctionCallItems(messages.flatMap((m) => responsesInputItemsFromMessage(m as never))); + + const callIds = items.filter((i) => i.type === "function_call").map((i) => i.call_id); + const outputIds = items.filter((i) => i.type === "function_call_output").map((i) => i.call_id); + assert.deepEqual( + [...callIds].sort(), + [...outputIds].sort(), + "every surviving function_call must have exactly one matching function_call_output", + ); + assert.ok(callIds.length > 0, "expected some paired tool calls to survive"); + }); + + it("a tight-budget trimmed session also yields strictly paired items", () => { + const messages: ApiMessage[] = [ + { role: "system", content: "sys" }, + { role: "user", content: "go" }, + ...toolCallGroup("call_a", "a.ts", 3000), + { role: "user", content: `more ${"z".repeat(4000)}` }, + ...toolCallGroup("call_b", "b.ts", 3000), + { role: "user", content: "final" }, + ]; + const trimmed = trimOldMessagesToFitContext(messages as never, 500, historyByteCapForBudget(500)); + void trimmed; + const items = pairResponsesFunctionCallItems(messages.flatMap((m) => responsesInputItemsFromMessage(m as never))); + const calls = items.filter((i) => i.type === "function_call").map((i) => i.call_id); + const outs = items.filter((i) => i.type === "function_call_output").map((i) => i.call_id); + assert.deepEqual([...calls].sort(), [...outs].sort()); + }); +}); + +describe("#217: luna Responses event shapes extract into parts", () => { + before(async () => { + const extractors = await import("../transports/extractors.js"); + OpenAiResponseExtractor = extractors.OpenAiResponseExtractor; + const routing = await import("../core/routing.js"); + normalizeResponsesStreamEvent = routing.normalizeResponsesStreamEvent; + }); + + it("tool-call stream emits a complete tool call despite zero text", () => { + const extractor = new OpenAiResponseExtractor(undefined, undefined, undefined, undefined, undefined, undefined, false); + const events = lunaToolCallEvents("call_QijniLAwknJ2xoh0Ak3IQn9A", ['{"', "filePath", '":"', "/x.ts", '"}']); + const parts: Array<{ name?: string; input?: unknown }> = []; + for (const event of events) { + for (const part of extractor.extractStreamParts(normalizeResponsesStreamEvent(event))) { + parts.push(part as { name?: string; input?: unknown }); + } + } + // The response.completed event carries finish_reason, which is where the + // extractor flushes accumulated tool calls. + const tool = parts.find((p) => typeof p.name === "string"); + assert.ok(tool, "expected at least one usable part from a healthy luna tool-call stream"); + assert.equal(tool.name, "read_file"); + assert.deepEqual(tool.input, { filePath: "/x.ts" }); + }); + + it("text stream emits text for flat and nested output_text.delta shapes", () => { + const extractor = new OpenAiResponseExtractor(undefined, undefined, undefined, undefined, undefined, undefined, false); + const events: Array> = [ + { type: "response.output_text.delta", delta: "Hello" }, + { type: "response.output_text.delta", delta: { text: " nested" } }, + { type: "response.output_text.delta", text: { value: " world" } }, + { + type: "response.completed", + response: { id: "resp_y", status: "completed", stop_reason: "stop", usage: { input_tokens: 1, output_tokens: 1 } }, + }, + ]; + let text = ""; + for (const event of events) { + for (const part of extractor.extractStreamParts(normalizeResponsesStreamEvent(event))) { + // LanguageModelTextPart carries the text in `value`. + const asText = part as { value?: string; text?: string }; + if (typeof asText.value === "string") text += asText.value; + else if (typeof asText.text === "string") text += asText.text; + } + } + assert.match(text, /Hello/); + assert.match(text, /nested/); + assert.match(text, /world/); + }); +}); diff --git a/src/test/responsesRequest.test.ts b/src/test/responsesRequest.test.ts index bdccb2e1c..7f4b8f079 100644 --- a/src/test/responsesRequest.test.ts +++ b/src/test/responsesRequest.test.ts @@ -1,6 +1,6 @@ import assert from "node:assert/strict"; import { describe, it } from "node:test"; -import { buildResponsesRequestEnvelope, responsesInputItemsFromMessage } from "../responsesRequest.js"; +import { buildResponsesRequestEnvelope, pairResponsesFunctionCallItems, responsesInputItemsFromMessage } from "../responsesRequest.js"; describe("buildResponsesRequestEnvelope", () => { it("enables server-side input truncation for long Responses sessions", () => { @@ -154,4 +154,48 @@ describe("responsesInputItemsFromMessage", () => { }); assert.deepEqual(items, []); }); + + it("emits no function_call_output when tool_call_id is missing (issue #216)", () => { + const items = responsesInputItemsFromMessage({ + role: "tool", + content: "result without an id", + }); + assert.deepEqual(items, []); + }); +}); + +describe("pairResponsesFunctionCallItems (issue #216)", () => { + it("keeps a matched function_call/function_call_output pair", () => { + const paired = pairResponsesFunctionCallItems([ + { type: "function_call", id: "fc_a", call_id: "call_a", name: "f", arguments: "{}" }, + { type: "function_call_output", call_id: "call_a", output: "ok" }, + ]); + assert.equal(paired.length, 2); + }); + + it("drops a function_call whose output was trimmed", () => { + const paired = pairResponsesFunctionCallItems([ + { type: "function_call", id: "fc_a", call_id: "call_a", name: "f", arguments: "{}" }, + { role: "user", content: "hi" }, + ]); + assert.deepEqual(paired, [{ role: "user", content: "hi" }]); + }); + + it("drops an output whose function_call was trimmed", () => { + const paired = pairResponsesFunctionCallItems([ + { type: "function_call_output", call_id: "call_a", output: "orphan" }, + { role: "user", content: "hi" }, + ]); + assert.deepEqual(paired, [{ role: "user", content: "hi" }]); + }); + + it("keeps only the first output for a duplicated call_id", () => { + const paired = pairResponsesFunctionCallItems([ + { type: "function_call", id: "fc_a", call_id: "call_a", name: "f", arguments: "{}" }, + { type: "function_call_output", call_id: "call_a", output: "first" }, + { type: "function_call_output", call_id: "call_a", output: "second" }, + ]); + assert.equal(paired.length, 2); + assert.equal((paired[1] as { output: string }).output, "first"); + }); }); diff --git a/src/test/session-header.test.ts b/src/test/session-header.test.ts new file mode 100644 index 000000000..72dd2aa5d --- /dev/null +++ b/src/test/session-header.test.ts @@ -0,0 +1,135 @@ +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { installVscodeMock } from "./helpers/goUsageTestUtils.js"; + +/* + * Gateway enforcement (docs/go, updated 2026-09-07): EVERY request to the + * OpenCode gateway must carry a stable x-opencode-session. The main chat path + * always did; these tests pin the auxiliary fetch sites that previously hit + * the gateway without the header: + * - GET /models (ModelListFetcher) + * - inline completions POST (ChatCompletionEngine) + * - GET /usage (fetchGoUsage) + */ + +installVscodeMock(); + +interface CapturedRequest { + url: string; + headers: Record; +} + +function stubFetch(capture: CapturedRequest[], respond: () => Response): void { + globalThis.fetch = ((input: unknown, init?: { headers?: Record }) => { + capture.push({ + url: String(input), + headers: normalizeHeaders(init?.headers), + }); + return Promise.resolve(respond()); + }) as unknown as typeof fetch; +} + +function normalizeHeaders(headers: unknown): Record { + const out: Record = {}; + if (!headers) return out; + if (headers instanceof Headers) { + headers.forEach((value, key) => { + out[key.toLowerCase()] = value; + }); + return out; + } + for (const [key, value] of Object.entries(headers as Record)) { + out[key.toLowerCase()] = String(value); + } + return out; +} + +function fakeContext(existing?: string): { + globalState: { get(key: string): unknown; update(key: string, value: unknown): Thenable }; +} { + const store = new Map(existing ? [["opencode.auxSessionId", existing]] : []); + return { + globalState: { + get: (key: string) => store.get(key), + update: (key: string, value: unknown) => { + store.set(key, value); + return Promise.resolve(); + }, + }, + }; +} + +describe("x-opencode-session header on auxiliary gateway requests", () => { + it("ModelListFetcher sends the session header on GET /models", async () => { + const { ModelListFetcher } = await import("../provider/modelList.js"); + const capture: CapturedRequest[] = []; + stubFetch( + capture, + () => new Response(JSON.stringify({ data: [{ id: "model-1" }] }), { status: 200, headers: { "content-type": "application/json" } }), + ); + + const fetcher = new ModelListFetcher({ + context: fakeContext() as never, + definition: { + vendor: "opencodego", + displayName: "OpenCode Go", + modelsUrl: "https://opencode.ai/zen/go/v1/models", + fallbackModels: [], + } as never, + log: () => {}, + replaceLiveModelMetadata: () => {}, + filterAvailableModels: (ids: string[]) => Promise.resolve(ids), + }); + const ids = await fetcher.fetch("sk-test"); + assert.deepEqual(ids, ["model-1"]); + assert.equal(capture.length, 1); + const session = capture[0].headers["x-opencode-session"]; + assert.ok( + session && session.startsWith("vscode-aux-"), + `expected auxiliary session header, got: ${JSON.stringify(capture[0].headers)}`, + ); + }); + + it("auxiliary session id is stable across calls and persisted in globalState", async () => { + const { auxiliarySessionId } = await import("../request/headers.js"); + const context = fakeContext(); + const first = auxiliarySessionId(context as never); + const second = auxiliarySessionId(context as never); + assert.ok(first.startsWith("vscode-aux-")); + assert.equal(first, second, "session id must be stable for the same installation state"); + assert.equal(context.globalState.get("opencode.auxSessionId"), first, "session id must be persisted to globalState"); + // A different installation (empty state) gets its own id. + assert.notEqual(auxiliarySessionId(fakeContext() as never), first); + }); + + it("ChatCompletionEngine sends the session header on inline completions", async () => { + const { ChatCompletionEngine } = await import("../autocomplete/engine.js"); + const capture: CapturedRequest[] = []; + stubFetch( + capture, + () => + new Response('data: {"choices":[{"delta":{"content":"ok"}}]}\n\ndata: [DONE]\n\n', { + status: 200, + headers: { "content-type": "text/event-stream" }, + }), + ); + + const engine = new ChatCompletionEngine({ + chatCompletionsUrl: "https://opencode.ai/zen/go/v1/chat/completions", + apiKey: "sk-test", + sessionId: "vscode-aux-test", + }); + await engine.complete({ prefix: "const x = ", suffix: "", maxTokens: 64, modelId: "test-model" }, new AbortController().signal); + assert.equal(capture.length, 1); + assert.equal(capture[0].headers["x-opencode-session"], "vscode-aux-test"); + }); + + it("fetchGoUsage sends the session header when a session id is supplied", async () => { + const { fetchGoUsage } = await import("../usage/goUsageSync.js"); + const capture: CapturedRequest[] = []; + stubFetch(capture, () => new Response(JSON.stringify({ ok: true }), { status: 200, headers: { "content-type": "application/json" } })); + await fetchGoUsage("sk-test", undefined, 1000, "https://opencode.ai/zen/go/v1/usage", "vscode-aux-usage"); + assert.equal(capture.length, 1); + assert.equal(capture[0].headers["x-opencode-session"], "vscode-aux-usage"); + }); +}); diff --git a/src/transports/engine.ts b/src/transports/engine.ts index 260b5a824..66d79a98e 100644 --- a/src/transports/engine.ts +++ b/src/transports/engine.ts @@ -15,7 +15,12 @@ import { TRANSIENT_5XX_RETRY_BASE_MS, TRANSIENT_5XX_RETRY_JITTER_MS, } from "../retry"; -import { TRANSIENT_FETCH_MAX_RETRIES, TRANSIENT_FETCH_RETRY_BASE_MS, TRANSIENT_FETCH_RETRY_JITTER_MS } from "../config"; +import { + TRANSIENT_FETCH_MAX_RETRIES, + TRANSIENT_FETCH_RETRY_BASE_MS, + TRANSIENT_FETCH_RETRY_JITTER_MS, + RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS, +} from "../config"; import { createUsageDataParts } from "../chatParts"; import { clearContextWindowRequest, @@ -23,7 +28,7 @@ import { setContextWindowOutputBufferForRequest, } from "../contextWindowHookBridge"; import { formatUsageLogLine } from "../usage/usage"; -import { getErrorMessage, sleepWithCancellation } from "../utils"; +import { getErrorMessage, parseRetryAfterMs, sleepWithCancellation } from "../utils"; import { parseServerSentEvent, isStreamTruncated } from "./sse"; import { reportProgressPart, type RequestUsageSummary, type StreamOpenCodeResponseOptions } from "./streamParts"; import type { TransportRequestSummary } from "../core/transport"; @@ -51,6 +56,17 @@ const STREAM_FAILURE_MAX_RETRIES = 3; */ const MAX_400_PATCH_ATTEMPTS = 3; +/** + * Read the token flag through a function boundary: aliased-condition narrowing + * on `options.token.isCancellationRequested` (from earlier guards in + * streamOpenCodeResponse) makes direct reads look constant-false to + * `no-unnecessary-condition` even after an await, where cancellation can + * actually fire. + */ +function cancellationRequested(token: { readonly isCancellationRequested: boolean }): boolean { + return token.isCancellationRequested; +} + /** * Core streaming engine shared by every transport: performs the HTTP POST * (with HTTP-400 body patching and transient-5xx backoff retries), parses the @@ -318,6 +334,24 @@ export async function streamOpenCodeResponse(options: StreamOpenCodeResponseOpti throw new DOMException("Aborted", "AbortError"); } + // --- 429 Retry-After retry (issue #221) --- + // Upstream provider rate limits (Console Go) often carry Retry-After. + // Honor it with a single bounded wait and one retry. Nothing has been + // streamed at this point, so the retry cannot duplicate chat content. + // Waits longer than RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS are surfaced as + // the normal 429 error instead of silently stalling the UI. + for (let rateLimitAttempt = 0; rateLimitAttempt < 1; rateLimitAttempt++) { + if (response.status !== 429 || options.token.isCancellationRequested) break; + const retryAfter = parseRetryAfterMs(response.headers.get("retry-after")); + if (retryAfter === undefined || retryAfter > RATE_LIMIT_MAX_RETRY_AFTER_WAIT_MS) break; + options.output?.appendLine(`[retry] 429 rate limited; honoring Retry-After, waiting ${String(retryAfter)}ms…`); + await sleepWithCancellation(retryAfter, options.token); + if (cancellationRequested(options.token)) break; + response = await fetchWithTransientRetry(payload); + consumedErrorBody = undefined; + options.output?.appendLine(`[retry] Response after rate-limit wait: ${String(response.status)} ${response.statusText}`); + } + responseStatus = response.status; responseContentType = response.headers.get("content-type") ?? ""; options.output?.appendLine(`[http] ${String(response.status)} ${response.statusText} content-type=${responseContentType || ""}`); @@ -457,7 +491,20 @@ export async function streamOpenCodeResponse(options: StreamOpenCodeResponseOpti // extractor found nothing, dump raw SSE data to identify format mismatches. // This helps diagnose issues like #93 where the model generates tokens // but the response content is in an unrecognized format. - if (usageSummary.completionTokens && usageSummary.completionTokens > 0 && extractedPartCount === 0 && rawSseData.length > 0) { + // + // ISSUE #217 follow-up: skip the dump when a healthy finish_reason was + // extracted. Tool-call responses on the Responses API (gpt-5.6-luna) emit + // parts via the transport's finally flush AFTER this block runs, so + // extractedPartCount is legitimately 0 here while the stream is perfectly + // healthy — the dump would false-positive on every tool-call turn. + const healthyFinish = typeof usageSummary.finishReason === "string" && usageSummary.finishReason !== "error"; + if ( + usageSummary.completionTokens && + usageSummary.completionTokens > 0 && + extractedPartCount === 0 && + !healthyFinish && + rawSseData.length > 0 + ) { options.output?.appendLine( `[diag-empty-response] model=${options.modelId} completionTokens=${String(usageSummary.completionTokens)} totalEvents=${String(totalEvents)} rawSseDataCount=${String(rawSseData.length)}`, ); @@ -525,6 +572,20 @@ export async function streamOpenCodeResponse(options: StreamOpenCodeResponseOpti }); return; } + // ISSUE #217: retries exhausted with zero usable parts while the + // gateway billed completion tokens — give the reporter a targeted + // message that names the likely cause and the diagnostic path. + if (extractedPartCount === 0 && usageSummary.completionTokens && usageSummary.completionTokens > 0) { + const zeroPartError = new OpenCodeRequestError( + `${options.providerDisplayName} consumed ${String(usageSummary.completionTokens)} completion tokens but returned no parsable content (${String(totalEvents)} events, ${String(totalBytes)} bytes${localRequestId ? `, request ${localRequestId}` : ""}).`, + `${options.providerDisplayName} generated output in a format this extension could not parse. Enable "Debug Reasoning" for this provider, retry once (the upstream may recover), and if it repeats, attach the [diag-sse-event-*] lines from the OpenCode Output channel to a bug report.`, + ); + emitSummary(totalBytes, totalEvents, { + errorMessage: zeroPartError.message, + rateLimitSummary, + }); + throw zeroPartError; + } const requestError = new OpenCodeRequestError( `${options.providerDisplayName} response stream ended before completion (no [DONE] or finish_reason after ${String(totalBytes)} bytes / ${String(totalEvents)} events${localRequestId ? `, request ${localRequestId}` : ""}).`, `${options.providerDisplayName} stopped sending data before the response was complete (the connection closed unexpectedly). Your message may be cut off — try sending it again; a single resend usually succeeds. If this keeps happening, check your connection, VPN, or firewall.`, diff --git a/src/usage/goUsageSync.ts b/src/usage/goUsageSync.ts index 46521709b..903f29b69 100644 --- a/src/usage/goUsageSync.ts +++ b/src/usage/goUsageSync.ts @@ -65,6 +65,7 @@ export async function fetchGoUsage( fetcher: typeof fetch = fetch, timeoutMs: number = GO_USAGE_FETCH_TIMEOUT_MS, endpointUrl: string = GO_USAGE_API_URL, + sessionId?: string, ): Promise { if (!apiKey) { return { ok: false, reason: "no-key" }; @@ -73,7 +74,13 @@ export async function fetchGoUsage( try { response = await fetcher(endpointUrl, { method: "GET", - headers: { Authorization: `Bearer ${apiKey}` }, + headers: { + Authorization: `Bearer ${apiKey}`, + // Gateway enforcement (docs/go): auxiliary OpenCode requests also need + // a stable session id. No conversation here — caller supplies the + // persisted per-installation id. + ...(sessionId ? { "x-opencode-session": sessionId } : {}), + }, signal: AbortSignal.timeout(timeoutMs), }); } catch { diff --git a/src/usage/tracker.ts b/src/usage/tracker.ts index 35825c1d1..6ac0d807d 100644 --- a/src/usage/tracker.ts +++ b/src/usage/tracker.ts @@ -2,6 +2,7 @@ import * as vscode from "vscode"; import type { ModelCost } from "../models/metadata"; import type { TransportRequestSummary } from "../core/transport"; import { fetchGoUsage, mergeServerUsage, GO_USAGE_SYNC_TTL_MS, type GoUsageApiResponse } from "./goUsageSync"; +import { auxiliarySessionId } from "../request/headers"; import { GO_LIMITS, GO_USAGE_LOG_KEY, @@ -352,7 +353,7 @@ export class GoUsageTracker { if (this.serverUsageFetchedAt > 0 && now - this.serverUsageFetchedAt < GO_USAGE_SYNC_TTL_MS) { return false; } - const result = await fetchGoUsage(apiKey, fetch, undefined, this.options.resolveUsageUrl?.()); + const result = await fetchGoUsage(apiKey, fetch, undefined, this.options.resolveUsageUrl?.(), auxiliarySessionId(this.context)); // Pace retries after failures too — an invalid key or unreachable // endpoint must not hammer the API on every request. this.serverUsageFetchedAt = Date.now(); diff --git a/src/utils.ts b/src/utils.ts index 1a3832576..e535256dd 100644 --- a/src/utils.ts +++ b/src/utils.ts @@ -186,3 +186,21 @@ export function sleepWithCancellation(ms: number, token: LikeCancellationToken): } }); } + +/** + * Parse a Retry-After header value into milliseconds. Accepts delta-seconds + * ("2") and HTTP-dates; returns undefined for absent or malformed values. + * Callers are expected to cap the result (issue #221). + */ +export function parseRetryAfterMs(value: string | null): number | undefined { + if (!value) return undefined; + const seconds = Number(value); + if (Number.isFinite(seconds) && seconds >= 0) { + return Math.round(seconds * 1000); + } + const at = Date.parse(value); + if (!Number.isNaN(at)) { + return Math.max(0, at - Date.now()); + } + return undefined; +}