From 4c317e07dc5246a38ef115a5785faf9e3473084b Mon Sep 17 00:00:00 2001 From: ltmoerdani Date: Fri, 25 Sep 2026 11:31:00 +0700 Subject: [PATCH 1/2] fix(responses): done-event repair must not clobber tool-call identity (#244) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 0.7.7 #244 repair delta forwarded id: firstString(call_id, item_id). The real luna function_call_arguments.done event carries only item_id (fc_1) — never call_id — so the accumulator adopted the item id, replacing the real call_* id captured from output_item.added. Item ids are reused across turns: two turns both carrying fc_1 produced two function_call items against one function_call_output at the gateway, which rejects the request with 400 "No tool output found for function call call_*". Three-layer fix: - done event forwards id only when a real call_id is present (arguments- only repair — the #244 arguments fix is preserved) - ToolCallAccumulator captures the id once, never overwritten by later fragments - pairResponsesFunctionCallItems collapses duplicate function_call items sharing a call_id, self-healing histories poisoned by 0.7.7 Proven empirically against the captured luna event shapes: part id stays call_* end to end. 4 regression tests; 493/493 pass. Version bump 0.7.8. Docs: docs/issues/108, doc 107 annotated, CHANGELOG 0.7.8, devlog, ARCHITECTURE-MAP. --- ARCHITECTURE-MAP.md | 4 +- CHANGELOG.md | 6 + docs/devlog.md | 16 +- ...issue244-tool-call-arguments-whitespace.md | 2 + ...issue244-followup-done-event-id-clobber.md | 60 +++++++ package-lock.json | 4 +- package.json | 2 +- src/core/routing.ts | 7 +- src/responsesRequest.ts | 16 +- src/test/issue244-followup-regression.test.ts | 170 ++++++++++++++++++ src/toolCallAccumulator.ts | 9 +- 11 files changed, 287 insertions(+), 9 deletions(-) create mode 100644 docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md create mode 100644 src/test/issue244-followup-regression.test.ts diff --git a/ARCHITECTURE-MAP.md b/ARCHITECTURE-MAP.md index 7cf56b2ca..2fb2993a3 100644 --- a/ARCHITECTURE-MAP.md +++ b/ARCHITECTURE-MAP.md @@ -37,7 +37,7 @@ Total `src/` ≈ **16,310 lines** across ~109 files (excl. tests). Grouped by do | --------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | **Provider** | `src/provider/` — `OpenCodeProvider.ts` (532), `chatPrep.ts` (332), `messages.ts` (446), `modelInfo.ts` (239), `historyTrim.ts` (200), `settings.ts` (260), `modelList.ts` (185), `visionProxy.ts` (345), `providerDialogs.ts` (117), `transportLog.ts` (109), `definitions.ts` (238), `tokens.ts` (106), `providerUtils.ts` (35) | ~3,017 | The `OpenCodeProvider` class implementing `LanguageModelChatProvider` (thin — delegates to siblings via deps objects), request preparation (`chatPrep`: message conversion, vision proxy, history trimming, budgets, thinking payload), model-info assembly (`modelInfo`), model-list fetch/cache (`modelList`), Manage/Test-Connection flows (`providerDialogs`), rolling transport diagnostics (`transportLog`) | `prepareChatRequest()` / `provideModelChatInformation()` / `convertMessage()` / `normalizeMessages()` / `trimOldMessagesToFitContext()` / `historyByteCapForBudget()` / `ModelListFetcher` | | **Transports** | `src/transports/` — `engine.ts` (409), `extractors.ts` (556), `extract.ts` (225), `thinkTags.ts` (139), `streamParts.ts` (88), `sse.ts` (31), `chatCompletions.ts` (64), `responses.ts` (30), `anthropic.ts` (28), `google.ts` (31) | ~1,601 | One adapter per wire format (OpenAI chat-completions, OpenAI Responses, Anthropic Messages, Google generateContent) + the shared streaming engine, SSE parser, response extractors | `streamOpenCodeResponse()` (engine), `OpenAiResponseExtractor` / `AnthropicResponseExtractor` | -| **Core (registry/routing)** | `src/core/` — `routing.ts` (535), `registry.ts` (142), `transport.ts` (70) | ~653 | Data-driven model registry (`MODEL_REGISTRY`), transport resolution (`resolveModelRouting`), Responses/Google SSE normalization, shared `StreamRequestOptions` contract | **Pure** — no `vscode` import, no side effects | +| **Core (registry/routing)** | `src/core/` — `routing.ts` (540), `registry.ts` (142), `transport.ts` (70) | ~653 | Data-driven model registry (`MODEL_REGISTRY`), transport resolution (`resolveModelRouting`), Responses/Google SSE normalization, shared `StreamRequestOptions` contract | **Pure** — no `vscode` import, no side effects | | **Models (metadata)** | `src/models/` — `metadata.ts` (523), `modelTables.ts` (141), `metadataFetcher.ts` (102), `modelLimits.ts` (52), `modelCapabilities.ts` (41), `modelNames.ts` (29), `pricing.ts` (88) | ~976 | models.dev live metadata + bundled fallback snapshot (static data tables in `modelTables.ts`), limit/capability resolution, pricing | Live fetch may fail → bundled snapshot MUST exist | | **Usage** | `src/usage/` — `tracker.ts` (585, thin class shell), `trackerTypes.ts` (96), `trackerWindows.ts` (140), `trackerSummary.ts` (238), `dashboard.ts` (19 barrel) + `dashboard/` (`webview.ts`, `webviewData.ts`, `webviewHtml.ts`, `state.ts`, `statusBar.ts`, `targetEditor.ts`, `tooltip.ts` ≈ 1,405), `history.ts` (378), `usage.ts` (146), `goUsageSync.ts` (128), `formatting.ts` (129), `usageProfile.ts` (74), `pricing.ts` (62) | ~3,571 | Go usage tracker (types/windows/summary split out of the old god file), per-profile tracking, CLI SQLite history reader, server-usage sync, status bar + usage webview + quick-pick (webview split into state/status/webview modules) | Server meters authoritative for Session/Weekly/Monthly; device-local for Today/Yesterday | | **Thinking** | `src/thinking/` — `provider.ts` (78), `base.ts` (75), `resolve.ts` (94), `deepseek.ts` (53), `glm.ts` (53), `kimi.ts` (81), `minimax.ts` (54), `mimo.ts` (72), `openai.ts` (57), `qwen.ts` (103), `fallback.ts` (39), `schema.ts` (106), `payload.ts` (26), `types.ts` (51) | ~942 | Per-family thinking strategy classes + config resolution (per-model config wins over workspace — but schema-default echoes are stripped first, `stripSchemaDefaultEcho` in `resolve.ts`, issue #226) | **Pure** — no `vscode` import; family from registry | @@ -45,7 +45,7 @@ Total `src/` ≈ **16,310 lines** across ~109 files (excl. tests). Grouped by do | **Commands** | `src/commands/` — `agentsWindow.ts` (130), `diagnostics.ts` (41), `providers.ts` (72), `thinkingPicker.ts` (31) | ~274 | Command handlers: diagnostics, agents-window BYOK bridge, provider enable/disable (state-aware toggle, base-vendor resolved — issue #228), thinking picker | Thin — delegates to provider/usage modules | | **Autocomplete** | `src/autocomplete/` — `index.ts` (157), `engine.ts` (143), `provider.ts` (127), `context.ts` (89), `usage.ts` (88), `throttle.ts` (79), `prompt.ts` (60), `types.ts` (32) | ~775 | Inline code suggestions (opt-in) — FIM emulation over chat-completions, debounce/throttle, usage counters | Separate subsystem; not wired into Go cost tracker yet | | **Extension entry** | `src/extension.ts` | 415 | Thin `activate()`/`deactivate()` — wiring only (command registration, provider registration, status bar init) | Target <300 lines; keep wiring-only | -| **Root utilities** | `src/config.ts` (278), `contextWindowHook.ts` (485), `contextWindowHookBridge.ts` (122), `errors.ts` (294), `retry.ts` (463), `utils.ts` (186), `responsesRequest.ts` (180), `toolCallAccumulator.ts` (180), `imageNormalizer.ts` (118), `visionProxyCache.ts` (79), `runtimeDiagnostics.ts` (57), `chatParts.ts` (55), `reasoningHistory.ts` (43), `tokenEstimate.ts` (39), `providerTypes.ts` (23), `openCodeAuth.ts` (20), `providerEnablement.ts` (18), `apiKeyResolution.ts` (8), `thinking.ts` (30, legacy barrel) | ~2,428 | Cross-cutting utilities (plus the two proposed-API `.d.ts` module augmentations ≈ 231 LoC — `chatProvider` v6 + `languageModelThinkingPart` v1) | `config.ts` must stay **dependency-free** | +| **Root utilities** | `src/config.ts` (278), `contextWindowHook.ts` (485), `contextWindowHookBridge.ts` (122), `errors.ts` (294), `retry.ts` (463), `utils.ts` (186), `responsesRequest.ts` (271), `toolCallAccumulator.ts` (187), `imageNormalizer.ts` (118), `visionProxyCache.ts` (79), `runtimeDiagnostics.ts` (57), `chatParts.ts` (55), `reasoningHistory.ts` (43), `tokenEstimate.ts` (39), `providerTypes.ts` (23), `openCodeAuth.ts` (20), `providerEnablement.ts` (18), `apiKeyResolution.ts` (8), `thinking.ts` (30, legacy barrel) | ~2,428 | Cross-cutting utilities (plus the two proposed-API `.d.ts` module augmentations ≈ 231 LoC — `chatProvider` v6 + `languageModelThinkingPart` v1) | `config.ts` must stay **dependency-free** | --- diff --git a/CHANGELOG.md b/CHANGELOG.md index 0d3cbcb95..724217996 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,12 @@ All notable changes to the **OpenCode Go BYOK Provider** extension are documented here. +## [0.7.8] — 2026-09-25 + +### Fixed + +- **`[Responses]` The #244 done-event repair no longer corrupts tool-call identity — the 0.7.7 regression causing `400 No tool output found for function call call_*` is fixed (#244 follow-up).** The real luna `response.function_call_arguments.done` event carries only `item_id` (`fc_1`), never `call_id`. The repair delta built `id: firstString(call_id, item_id)` and the accumulator adopted it, REPLACING the real `call_*` id captured from `output_item.added` — and item ids are reused across turns. Two turns both carrying `fc_1` produced two `function_call` items against one `function_call_output` at the gateway, which rejects the whole request with `No tool output found`. Fix: the done event now repairs **arguments only** (id forwarded only when a real `call_id` is present); `ToolCallAccumulator` captures an id once and never lets a later fragment overwrite it; and `pairResponsesFunctionCallItems` collapses duplicate `function_call` items sharing a call_id, self-healing histories already poisoned by 0.7.7. Proven empirically against the captured luna event shapes: part id stays `call_*` end to end. Documented in `docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md`. + ## [0.7.7] — 2026-09-24 ### Fixed diff --git a/docs/devlog.md b/docs/devlog.md index c0501dc5c..e88482d2c 100644 --- a/docs/devlog.md +++ b/docs/devlog.md @@ -1,6 +1,20 @@ # 🧠 OPENCODE COPILOT CHAT DEVLOG -**Branch:** `main` (fix uncommitted, branch TBD) | **Updated:** 2026-09-24 Asia/Jakarta | **Current Phase:** issue #244 fix — tool-call arguments whitespace on Responses API; implemented + tested (489/489), docs synced, CHANGELOG `[0.7.7]`, VSIX 0.7.7 built + installed locally, pending commit. +**Branch:** `fix/issue244-followup-id-clobber` (work on `main`) | **Updated:** 2026-09-25 Asia/Jakarta | **Current Phase:** #244 follow-up — 0.7.7 done-event repair clobbered tool-call identity (`400 No tool output found`); fixed in 3 layers, 493/493 tests, 0.7.8 pending release. + +--- + +## ✅ #244 Follow-up — done-event id clobber → 400 No tool output found — 2026-09-25 + +**Scope:** regression OF the #244 fix (0.7.7), reported by the same user 18h later: every OpenAI-model request 400s with `No tool output found for function call call_*` (#216 error class). The done-event repair delta built `id: firstString(call_id, item_id)` — but the real luna done event carries only `item_id: "fc_1"`, which the accumulator adopted, REPLACING the real `call_*` id. Item ids are reused across turns → two turns with `fc_1` = two function_calls against one output at the gateway → 400. Proven empirically by replaying the captured luna event shapes (part id came out `fc_1` instead of `call_Vf1vzJwf...`). + +**Fix (3 layers):** (1) done event forwards `id` only when a real `call_id` is present — repair is arguments-only; (2) `ToolCallAccumulator` captures id once, never overwritten; (3) `pairResponsesFunctionCallItems` collapses duplicate function_call items sharing a call_id — poisoned histories self-heal. + +**Tests:** 4 new in `src/test/issue244-followup-regression.test.ts` (identity preserved, round-trip pairing, #244 arguments repair preserved, poisoned-history dedupe). **493/493 pass**, compile clean. + +**Lesson:** a streaming repair event must never mutate call identity; `firstString(call_id, item_id)`-style fallbacks across semantically different identifiers are the trap. + +Docs: `docs/issues/108`, doc 107 annotated, CHANGELOG `[0.7.8]`. --- diff --git a/docs/issues/107-20260924-issue244-tool-call-arguments-whitespace.md b/docs/issues/107-20260924-issue244-tool-call-arguments-whitespace.md index 0cdd3c27c..cbe9c312e 100644 --- a/docs/issues/107-20260924-issue244-tool-call-arguments-whitespace.md +++ b/docs/issues/107-20260924-issue244-tool-call-arguments-whitespace.md @@ -54,6 +54,8 @@ The reporter's OpenCode Diagnostics dumps (Go + Zen) corroborate the root cause: Design note: the done-event path makes the fix **permanent** — even if some gateway trims or mangles delta fragments, the final `response.function_call_arguments.done` value overwrites the accumulated string, matching official SDK semantics (`output.arguments = event.arguments`). +> **⚠️ Follow-up regression (fixed in 0.7.8):** the original repair delta also forwarded `id: firstString(call_id, item_id)` — and the real luna done event carries only `item_id` (`fc_1`), which the accumulator adopted, clobbering the real `call_*` identity and causing `400 No tool output found` (item ids are reused across turns). See [doc 108](108-20260925-issue244-followup-done-event-id-clobber.md). The arguments-only repair remains; identity is never touched. + ## Tests - `src/test/routing.test.ts` — "function_call_arguments whitespace preservation (#244)": fragments split inside a JSON string value must preserve the space; done-event mapping; done-event without arguments emits no choices. diff --git a/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md b/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md new file mode 100644 index 000000000..91ce7334b --- /dev/null +++ b/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md @@ -0,0 +1,60 @@ +# Issue #244 Follow-up — 0.7.7 Done-Event Repair Corrupted Tool-Call Identity (`400 No tool output found`) + +**Status:** ✅ Solved (0.7.8) +**Topic:** streaming / responses-api / tool-calls / pairing +**Updated:** 2026-09-25 +**Tags:** #responses-api #bug #tool-calls #pairing #regression +**GitHub Issue:** [#244](https://github.com/ltmoerdani/opencode-copilot-chat/issues/244) (comment 18h after the 0.7.7 fix) +**Related:** doc [107 — #244 arguments whitespace](107-20260924-issue244-tool-call-arguments-whitespace.md) (the fix this regressed), docs [96 — #216 pairing](../issues/96-20260808-issue216-no-tool-output-for-function-call.md), [90 — #206 fc_ ids](90-20260903-issue206-luna-responses-fc-id-mismatch.md) + +--- + +## Problem + +Immediately after updating to 0.7.7, the reporter's every OpenAI-model request failed: + +```text +OpenCode Zen API request failed (400) model=gpt-5.6-luna payloadBytes=19484: +No tool output found for function call call_Vf1vzJwf5xa44CfwtrzYe4y7. +``` + +Same on OpenCode Go (`gpt-6-luna`, `call_d0KxiJXE7jjKAM4MQrXdbCZl`). All requests 400 at the gateway, before streaming. + +## Root Cause — the done-event repair changed tool-call IDENTITY + +The #244 fix (0.7.7) added a `response.function_call_arguments.done` handler emitting a repair delta tagged `argumentsDone: true`, with: + +```ts +id: firstString(data.call_id, data.item_id) ?? "", +``` + +The captured REAL luna event shapes (#216/#217 suite) show the done event carries **only `item_id: "fc_1"` — never `call_id`**. So the repair delta adopted `fc_1`, and `ToolCallAccumulator.collect()` REPLACEd the real `call_*` id captured from `output_item.added`. + +Fatal because **item ids are per-response and reused across turns**. Two turns both carrying `fc_1` as the part id produce, at the gateway: `function_call(fc_1) ×2` + `function_call_output(fc_1) ×1` — our pairing's `consumedOutputs` drops the second output, so the second function_call reaches the gateway unpaired → the exact 400. + +Proven empirically: replaying the captured luna event sequence through `normalizeResponsesStreamEvent` + `OpenAiResponseExtractor` emitted `callId: "fc_1"` instead of `call_Vf1vzJwf5xa44CfwtrzYe4y7`. + +## Fix + +| Layer | File | Change | +| ---------- | ---------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Root cause | `src/core/routing.ts` | Done event forwards `id` **only when a real `call_id` is present**; otherwise no id — the repair is arguments-only | +| Defense | `src/toolCallAccumulator.ts` | An id is captured **once** (first fragment that carries one) and never overwritten by later fragments | +| Self-heal | `src/responsesRequest.ts` | `pairResponsesFunctionCallItems` collapses duplicate `function_call` items sharing one call_id (keeps the first) — histories already poisoned by 0.7.7 recover automatically | + +The `argumentsDone` REPLACE repair for arguments (the actual #244 fix) is preserved and pinned by test. + +## Tests + +`src/test/issue244-followup-regression.test.ts` (renamed from the repro): + +1. Part id stays `call_*` when done carries only `item_id` (the exact reported shape). +2. Full round-trip: wire request pairs function_call with its output. +3. `argumentsDone` still repairs mis-joined arguments (#244 fix preserved). +4. Pairing collapses duplicate function_call items (poisoned-history self-heal). + +Verification: **493/493** tests pass, `npm run compile` clean. + +## Lesson Recorded + +A streaming **repair** event must never mutate call **identity** — identity comes from `output_item.added`, arguments from `*.delta`/`*.done`. When forwarding fields from a done event, forward only the fields the repair needs; `firstString(call_id, item_id)`-style fallbacks across semantically different identifiers are how `fc_` item ids leaked into call identity. diff --git a/package-lock.json b/package-lock.json index b1ad652cf..199e42d32 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "opencode-copilot-chat", - "version": "0.7.7", + "version": "0.7.8", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "opencode-copilot-chat", - "version": "0.7.7", + "version": "0.7.8", "license": "MIT", "dependencies": { "@silvia-odwyer/photon-node": "^0.3.4" diff --git a/package.json b/package.json index aadc7108b..e337b2336 100644 --- a/package.json +++ b/package.json @@ -2,7 +2,7 @@ "name": "opencode-copilot-chat", "displayName": "OpenCode for Copilot Chat: BYOK 30+ AI Models", "description": "Use 30+ frontier AI models (DeepSeek V4, Kimi K2.6, GLM-5.1, Qwen3.7, MiMo V2.5, MiniMax M2.7, free Claude Opus, GPT-5.5, Gemini 3.5, Grok) in GitHub Copilot Chat. Bring Your Own Key, no Copilot Pro needed.", - "version": "0.7.7", + "version": "0.7.8", "publisher": "ltmoerdani", "license": "MIT", "icon": "media/opencodego.png", diff --git a/src/core/routing.ts b/src/core/routing.ts index 5fef65d42..6677e6f42 100644 --- a/src/core/routing.ts +++ b/src/core/routing.ts @@ -152,11 +152,16 @@ export function normalizeResponsesStreamEvent(data: unknown): unknown { // Use it as an authoritative repair: the accumulator REPLACES (not appends) // the pending arguments so any corruption from mis-joined delta fragments is // healed at stream end (issue #244). + // IDENTITY RULE (0.7.7 regression): the done event may carry only `item_id` + // (e.g. `fc_1` on gpt-5.6-luna) — item ids are REUSED across turns, so they + // must never become the tool-call part id. Only a real `call_id` is + // forwarded; the accumulator keeps the id captured from output_item.added. if (eventType === "response.function_call_arguments.done") { const args = firstStringRaw(data.arguments); if (args === undefined) { return { choices: [] }; } + const callId = typeof data.call_id === "string" && data.call_id.trim() ? data.call_id : undefined; return { choices: [ { @@ -165,7 +170,7 @@ export function normalizeResponsesStreamEvent(data: unknown): unknown { tool_calls: [ { index: typeof data.output_index === "number" ? data.output_index : 0, - id: firstString(data.call_id, data.item_id) ?? "", + ...(callId ? { id: callId } : {}), type: "function", function: { arguments: args }, argumentsDone: true, diff --git a/src/responsesRequest.ts b/src/responsesRequest.ts index 3d47f55cf..3e546170f 100644 --- a/src/responsesRequest.ts +++ b/src/responsesRequest.ts @@ -137,6 +137,10 @@ export function responsesInputItemsFromMessage(message: ResponsesApiMessage): Re * - A `function_call` with no output is dropped too (the gateway 400s on it; * keeping it would fail the entire turn). * - The first output wins if a call_id is duplicated. + * - Duplicate `function_call` items sharing one call_id are collapsed to the + * first (self-heals histories poisoned by 0.7.7's id-clobbering bug, where + * reused `fc_*` item ids made two turns emit the same call id; #244 + * follow-up). */ export function pairResponsesFunctionCallItems(items: Record[]): Record[] { const callIdsWithOutput = new Set(); @@ -156,6 +160,7 @@ export function pairResponsesFunctionCallItems(items: Record[]) } } const consumedOutputs = new Set(); + const seenCalls = new Set(); return items.filter((item) => { const callId = item.call_id; if (item.type === "function_call_output") { @@ -164,7 +169,16 @@ export function pairResponsesFunctionCallItems(items: Record[]) return matched; } if (item.type === "function_call") { - return typeof callId === "string" && callIdsWithOutput.has(callId); + if (typeof callId !== "string" || !callIdsWithOutput.has(callId)) { + return false; + } + // Collapse duplicate function_call items: one output can only serve one + // call, so extras with the same call_id would 400 at the gateway. + if (seenCalls.has(callId)) { + return false; + } + seenCalls.add(callId); + return true; } return true; }); diff --git a/src/test/issue244-followup-regression.test.ts b/src/test/issue244-followup-regression.test.ts new file mode 100644 index 000000000..43ce1c647 --- /dev/null +++ b/src/test/issue244-followup-regression.test.ts @@ -0,0 +1,170 @@ +import { describe, it, before } from "node:test"; +import assert from "node:assert/strict"; +import Module from "node:module"; +import path from "node:path"; +import fs from "node:fs"; +import os from "node:os"; + +/* + * Regression tests for the 0.7.7 #244 follow-up: 400 "No tool output found + * for function call call_*". The argumentsDone repair delta carried + * `id: firstString(call_id, item_id)` — and the real luna done event has NO + * call_id, only `item_id: "fc_1"`. ToolCallAccumulator adopted that item id, + * REPLACING the real call_* id from output_item.added. Item ids are reused + * across turns, so two turns both carrying `fc_1` produced two function_call + * items with one output at the gateway → 400. + * + * Contract now: the done event repairs ARGUMENTS ONLY — it must never change + * tool-call identity. Pairing additionally collapses duplicate function_call + * items to self-heal already-poisoned histories. + */ + +const vscodeMockPath = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "vscode-mock-244b-")), "index.js"); +fs.writeFileSync( + vscodeMockPath, + `"use strict"; +class LanguageModelTextPart { constructor(value) { this.value = value; } } +class LanguageModelThinkingPart { constructor(text) { this.text = text; } } +class LanguageModelToolCallPart { constructor(callId, name, input) { this.callId = callId; this.name = name; this.input = input; } } +class LanguageModelToolResultPart { constructor(callId, content) { this.callId = callId; this.content = content; } } +module.exports = { LanguageModelTextPart, LanguageModelThinkingPart, LanguageModelToolCallPart, LanguageModelToolResultPart }; +`, + "utf-8", +); + +type ResolveFilename = (request: string, parent: unknown, ...args: unknown[]) => string; +const moduleResolver = Module as unknown as { _resolveFilename: ResolveFilename }; +const originalResolveFilename = moduleResolver._resolveFilename; +moduleResolver._resolveFilename = function (request, parent, ...args) { + if (request === "vscode") return vscodeMockPath; + return originalResolveFilename.call(this, request, parent, ...args); +}; + +let OpenAiResponseExtractor: typeof import("../transports/extractors.js").OpenAiResponseExtractor; +let normalizeResponsesStreamEvent: typeof import("../core/routing.js").normalizeResponsesStreamEvent; +let responsesInputItemsFromMessage: typeof import("../responsesRequest.js").responsesInputItemsFromMessage; +let pairResponsesFunctionCallItems: typeof import("../responsesRequest.js").pairResponsesFunctionCallItems; + +function lunaToolCallEvents(callId: string, argsChunks: string[]): Array> { + const events: Array> = [ + { type: "response.created", sequence_number: 0, response: { id: "resp_x", status: "in_progress", output: [], usage: null } }, + { + type: "response.output_item.added", + sequence_number: 4, + output_index: 1, + item: { id: "fc_1", type: "function_call", status: "in_progress", name: "read_file", call_id: callId, arguments: "" }, + }, + ]; + let seq = 5; + for (const chunk of argsChunks) { + events.push({ type: "response.function_call_arguments.delta", sequence_number: seq++, output_index: 1, item_id: "fc_1", delta: chunk }); + } + // NOTE: the real luna done event carries item_id only — NO call_id. + events.push({ + type: "response.function_call_arguments.done", + sequence_number: seq++, + output_index: 1, + item_id: "fc_1", + arguments: argsChunks.join(""), + }); + events.push({ + type: "response.output_item.done", + sequence_number: seq++, + output_index: 1, + item: { id: "fc_1", type: "function_call", status: "completed", name: "read_file", call_id: callId, arguments: argsChunks.join("") }, + }); + events.push({ + type: "response.completed", + sequence_number: seq++, + response: { id: "resp_x", status: "completed", stop_reason: "tool_calls", usage: { input_tokens: 100, output_tokens: 50 } }, + }); + return events; +} + +describe("#244 follow-up: done event repairs arguments, never identity", () => { + before(async () => { + const extractors = await import("../transports/extractors.js"); + OpenAiResponseExtractor = extractors.OpenAiResponseExtractor; + const routing = await import("../core/routing.js"); + normalizeResponsesStreamEvent = routing.normalizeResponsesStreamEvent; + const responses = await import("../responsesRequest.js"); + responsesInputItemsFromMessage = responses.responsesInputItemsFromMessage; + pairResponsesFunctionCallItems = responses.pairResponsesFunctionCallItems; + }); + + it("part id stays the call_* id when done carries only item_id", () => { + const extractor = new OpenAiResponseExtractor(undefined, undefined, undefined, undefined, undefined, undefined, false); + const parts: Array<{ callId?: string; name?: string; input?: unknown }> = []; + for (const event of lunaToolCallEvents("call_Vf1vzJwf5xa44CfwtrzYe4y7", ['{"', "filePath", '":"', "/x.ts", '"}'])) { + for (const part of extractor.extractStreamParts(normalizeResponsesStreamEvent(event))) { + parts.push(part as { callId?: string; name?: string; input?: unknown }); + } + } + console.log("EMITTED PARTS:", JSON.stringify(parts.map((p) => ({ callId: p.callId, name: p.name, input: p.input })))); + const tool = parts.find((p) => typeof p.name === "string"); + assert.ok(tool, "expected a tool call part"); + assert.equal(tool.callId, "call_Vf1vzJwf5xa44CfwtrzYe4y7", "part id must remain the gateway call_id, not the fc_ item id"); + }); + + it("full round-trip: wire request pairs function_call with its output", () => { + const extractor = new OpenAiResponseExtractor(undefined, undefined, undefined, undefined, undefined, undefined, false); + const parts: Array<{ callId?: string; name?: string; input?: unknown }> = []; + for (const event of lunaToolCallEvents("call_ROUNDTRIP", ['{"', "filePath", '":"', "/x.ts", '"}'])) { + for (const part of extractor.extractStreamParts(normalizeResponsesStreamEvent(event))) { + parts.push(part as { callId?: string; name?: string; input?: unknown }); + } + } + const tool = parts.find((p) => typeof p.name === "string") as { callId: string; name: string; input: object }; + const assistant = { + role: "assistant", + content: null, + tool_calls: [{ id: tool.callId, type: "function", function: { name: tool.name, arguments: JSON.stringify(tool.input) } }], + }; + const toolResult = { role: "tool", tool_call_id: tool.callId, content: "file body" }; + const items = pairResponsesFunctionCallItems([assistant, toolResult].flatMap((m) => responsesInputItemsFromMessage(m as never))); + const calls = items.filter((i) => i.type === "function_call").map((i) => i.call_id); + const outs = items.filter((i) => i.type === "function_call_output").map((i) => i.call_id); + assert.deepEqual([...calls].sort(), [...outs].sort(), "every function_call must have exactly one matching output"); + }); + + it("argumentsDone still repairs arguments (the #244 fix is preserved)", () => { + const extractor = new OpenAiResponseExtractor(undefined, undefined, undefined, undefined, undefined, undefined, false); + // Deltas mis-joined by a gateway, then the authoritative done value. + const events: Array> = [ + { + type: "response.output_item.added", + output_index: 0, + item: { id: "fc_9", type: "function_call", status: "in_progress", name: "search", call_id: "call_ZZZ", arguments: "" }, + }, + { type: "response.function_call_arguments.delta", output_index: 0, item_id: "fc_9", delta: '{"query":"what' }, + { type: "response.function_call_arguments.delta", output_index: 0, item_id: "fc_9", delta: " is 2 plus 2?" }, + { type: "response.function_call_arguments.done", output_index: 0, item_id: "fc_9", arguments: '{"query":"what is 2 plus 2?"}' }, + { type: "response.completed", response: { id: "r", status: "completed", stop_reason: "tool_calls", usage: null } }, + ]; + const parts: Array<{ callId?: string; name?: string; input?: unknown }> = []; + for (const event of events) { + for (const part of extractor.extractStreamParts(normalizeResponsesStreamEvent(event))) { + parts.push(part as { callId?: string; name?: string; input?: unknown }); + } + } + const tool = parts.find((p) => typeof p.name === "string") as { callId: string; input: object }; + assert.equal(tool.callId, "call_ZZZ", "id from output_item.added must survive"); + assert.deepEqual(tool.input, { query: "what is 2 plus 2?" }, "arguments must be the authoritative done value"); + }); + + it("pairing collapses duplicate function_call items sharing one call_id (poisoned history self-heal)", () => { + // History poisoned by 0.7.7: two turns both used item id fc_1 as the part id. + const msg = () => ({ + role: "assistant", + content: null, + tool_calls: [{ id: "fc_1", type: "function", function: { name: "read_file", arguments: '{"filePath":"x.ts"}' } }], + }); + const result = (content: string) => ({ role: "tool", tool_call_id: "fc_1", content }); + const items = [msg(), result("a"), msg(), result("b")].flatMap((m) => responsesInputItemsFromMessage(m as never)); + const paired = pairResponsesFunctionCallItems(items); + const calls = paired.filter((i) => i.type === "function_call"); + const outs = paired.filter((i) => i.type === "function_call_output"); + assert.equal(calls.length, 1, `expected 1 function_call after dedupe, got ${String(calls.length)}`); + assert.equal(outs.length, 1, `expected 1 function_call_output after dedupe, got ${String(outs.length)}`); + }); +}); diff --git a/src/toolCallAccumulator.ts b/src/toolCallAccumulator.ts index 9887b3fb7..30f336752 100644 --- a/src/toolCallAccumulator.ts +++ b/src/toolCallAccumulator.ts @@ -74,7 +74,14 @@ export class ToolCallAccumulator { name: "", arguments: "", }; - if (typeof toolCall.id === "string") { + // IDENTITY RULE (0.7.7 regression, #244 follow-up): an id is captured + // once (from the first fragment that carries one) and never overwritten + // by a later fragment. Late fragments may carry `item_id`-shaped ids + // (e.g. Responses `function_call_arguments.done` with only `fc_1`) that + // are reused across turns — adopting them corrupts the call id that + // history pairing and the gateway's function_call/_output matching + // depend on. + if (typeof toolCall.id === "string" && toolCall.id && !pending.id) { pending.id = toolCall.id; } From 370dc0886a5557b1045d2345c45ed5a39188ad0b Mon Sep 17 00:00:00 2001 From: ltmoerdani Date: Fri, 25 Sep 2026 13:17:09 +0700 Subject: [PATCH 2/2] fix(responses): enforce identity rule on every Responses entry point MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Post-fix audit found the #244-followup fallback (firstString(call_id, item_id)) in two more places: the output_item.added handler and the normalizeResponsesFullResponse function_call mapping. Both now derive call identity from call_id only — item ids (fc_*) are reused across turns and must never become call identity. Gateways that send call_id (all known real shapes) are unaffected. Adds 2 regression tests and a 17-check pre-release E2E simulating the full tool-call loop: whitespace preservation, wire pairing per turn, cross-turn item-id reuse, and poisoned-history self-heal. 495/495 unit tests, lint 7/7, retry E2E 9/9. --- ARCHITECTURE-MAP.md | 2 +- CHANGELOG.md | 2 +- docs/devlog.md | 2 +- ...issue244-followup-done-event-id-clobber.md | 25 ++++++++++++-- src/core/routing.ts | 10 ++++-- src/test/issue244-followup-regression.test.ts | 33 +++++++++++++++++++ 6 files changed, 67 insertions(+), 7 deletions(-) diff --git a/ARCHITECTURE-MAP.md b/ARCHITECTURE-MAP.md index 2fb2993a3..7178331fd 100644 --- a/ARCHITECTURE-MAP.md +++ b/ARCHITECTURE-MAP.md @@ -37,7 +37,7 @@ Total `src/` ≈ **16,310 lines** across ~109 files (excl. tests). Grouped by do | --------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | **Provider** | `src/provider/` — `OpenCodeProvider.ts` (532), `chatPrep.ts` (332), `messages.ts` (446), `modelInfo.ts` (239), `historyTrim.ts` (200), `settings.ts` (260), `modelList.ts` (185), `visionProxy.ts` (345), `providerDialogs.ts` (117), `transportLog.ts` (109), `definitions.ts` (238), `tokens.ts` (106), `providerUtils.ts` (35) | ~3,017 | The `OpenCodeProvider` class implementing `LanguageModelChatProvider` (thin — delegates to siblings via deps objects), request preparation (`chatPrep`: message conversion, vision proxy, history trimming, budgets, thinking payload), model-info assembly (`modelInfo`), model-list fetch/cache (`modelList`), Manage/Test-Connection flows (`providerDialogs`), rolling transport diagnostics (`transportLog`) | `prepareChatRequest()` / `provideModelChatInformation()` / `convertMessage()` / `normalizeMessages()` / `trimOldMessagesToFitContext()` / `historyByteCapForBudget()` / `ModelListFetcher` | | **Transports** | `src/transports/` — `engine.ts` (409), `extractors.ts` (556), `extract.ts` (225), `thinkTags.ts` (139), `streamParts.ts` (88), `sse.ts` (31), `chatCompletions.ts` (64), `responses.ts` (30), `anthropic.ts` (28), `google.ts` (31) | ~1,601 | One adapter per wire format (OpenAI chat-completions, OpenAI Responses, Anthropic Messages, Google generateContent) + the shared streaming engine, SSE parser, response extractors | `streamOpenCodeResponse()` (engine), `OpenAiResponseExtractor` / `AnthropicResponseExtractor` | -| **Core (registry/routing)** | `src/core/` — `routing.ts` (540), `registry.ts` (142), `transport.ts` (70) | ~653 | Data-driven model registry (`MODEL_REGISTRY`), transport resolution (`resolveModelRouting`), Responses/Google SSE normalization, shared `StreamRequestOptions` contract | **Pure** — no `vscode` import, no side effects | +| **Core (registry/routing)** | `src/core/` — `routing.ts` (546), `registry.ts` (142), `transport.ts` (70) | ~653 | Data-driven model registry (`MODEL_REGISTRY`), transport resolution (`resolveModelRouting`), Responses/Google SSE normalization, shared `StreamRequestOptions` contract | **Pure** — no `vscode` import, no side effects | | **Models (metadata)** | `src/models/` — `metadata.ts` (523), `modelTables.ts` (141), `metadataFetcher.ts` (102), `modelLimits.ts` (52), `modelCapabilities.ts` (41), `modelNames.ts` (29), `pricing.ts` (88) | ~976 | models.dev live metadata + bundled fallback snapshot (static data tables in `modelTables.ts`), limit/capability resolution, pricing | Live fetch may fail → bundled snapshot MUST exist | | **Usage** | `src/usage/` — `tracker.ts` (585, thin class shell), `trackerTypes.ts` (96), `trackerWindows.ts` (140), `trackerSummary.ts` (238), `dashboard.ts` (19 barrel) + `dashboard/` (`webview.ts`, `webviewData.ts`, `webviewHtml.ts`, `state.ts`, `statusBar.ts`, `targetEditor.ts`, `tooltip.ts` ≈ 1,405), `history.ts` (378), `usage.ts` (146), `goUsageSync.ts` (128), `formatting.ts` (129), `usageProfile.ts` (74), `pricing.ts` (62) | ~3,571 | Go usage tracker (types/windows/summary split out of the old god file), per-profile tracking, CLI SQLite history reader, server-usage sync, status bar + usage webview + quick-pick (webview split into state/status/webview modules) | Server meters authoritative for Session/Weekly/Monthly; device-local for Today/Yesterday | | **Thinking** | `src/thinking/` — `provider.ts` (78), `base.ts` (75), `resolve.ts` (94), `deepseek.ts` (53), `glm.ts` (53), `kimi.ts` (81), `minimax.ts` (54), `mimo.ts` (72), `openai.ts` (57), `qwen.ts` (103), `fallback.ts` (39), `schema.ts` (106), `payload.ts` (26), `types.ts` (51) | ~942 | Per-family thinking strategy classes + config resolution (per-model config wins over workspace — but schema-default echoes are stripped first, `stripSchemaDefaultEcho` in `resolve.ts`, issue #226) | **Pure** — no `vscode` import; family from registry | diff --git a/CHANGELOG.md b/CHANGELOG.md index 724217996..7177aa2cc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,7 +6,7 @@ All notable changes to the **OpenCode Go BYOK Provider** extension are documente ### Fixed -- **`[Responses]` The #244 done-event repair no longer corrupts tool-call identity — the 0.7.7 regression causing `400 No tool output found for function call call_*` is fixed (#244 follow-up).** The real luna `response.function_call_arguments.done` event carries only `item_id` (`fc_1`), never `call_id`. The repair delta built `id: firstString(call_id, item_id)` and the accumulator adopted it, REPLACING the real `call_*` id captured from `output_item.added` — and item ids are reused across turns. Two turns both carrying `fc_1` produced two `function_call` items against one `function_call_output` at the gateway, which rejects the whole request with `No tool output found`. Fix: the done event now repairs **arguments only** (id forwarded only when a real `call_id` is present); `ToolCallAccumulator` captures an id once and never lets a later fragment overwrite it; and `pairResponsesFunctionCallItems` collapses duplicate `function_call` items sharing a call_id, self-healing histories already poisoned by 0.7.7. Proven empirically against the captured luna event shapes: part id stays `call_*` end to end. Documented in `docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md`. +- **`[Responses]` The #244 done-event repair no longer corrupts tool-call identity — the 0.7.7 regression causing `400 No tool output found for function call call_*` is fixed (#244 follow-up).** The real luna `response.function_call_arguments.done` event carries only `item_id` (`fc_1`), never `call_id`. The repair delta built `id: firstString(call_id, item_id)` and the accumulator adopted it, REPLACING the real `call_*` id captured from `output_item.added` — and item ids are reused across turns. Two turns both carrying `fc_1` produced two `function_call` items against one `function_call_output` at the gateway, which rejects the whole request with `No tool output found`. Fix: the done event now repairs **arguments only** (id forwarded only when a real `call_id` is present); `ToolCallAccumulator` captures an id once and never lets a later fragment overwrite it; and `pairResponsesFunctionCallItems` collapses duplicate `function_call` items sharing a call_id, self-healing histories already poisoned by 0.7.7. The identity rule is enforced on every Responses entry point — `output_item.added` and the full-response mapping no longer fall back to `item.id` either. Proven empirically against the captured luna event shapes (part id stays `call_*` end to end) plus a 17-check E2E simulation of the full tool-call loop. Documented in `docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md`. ## [0.7.7] — 2026-09-24 diff --git a/docs/devlog.md b/docs/devlog.md index e88482d2c..aa1ae1fb4 100644 --- a/docs/devlog.md +++ b/docs/devlog.md @@ -1,6 +1,6 @@ # 🧠 OPENCODE COPILOT CHAT DEVLOG -**Branch:** `fix/issue244-followup-id-clobber` (work on `main`) | **Updated:** 2026-09-25 Asia/Jakarta | **Current Phase:** #244 follow-up — 0.7.7 done-event repair clobbered tool-call identity (`400 No tool output found`); fixed in 3 layers, 493/493 tests, 0.7.8 pending release. +**Branch:** `fix/issue244-followup-id-clobber` (work on `main`) | **Updated:** 2026-09-25 Asia/Jakarta | **Current Phase:** #244 follow-up fix complete + deep-verified (495/495 tests, lint 7/7, retry E2E 9/9, pre-release E2E 17/17); 0.7.8 pending push/PR/build. --- diff --git a/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md b/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md index 91ce7334b..49185f0c7 100644 --- a/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md +++ b/docs/issues/108-20260925-issue244-followup-done-event-id-clobber.md @@ -42,18 +42,39 @@ Proven empirically: replaying the captured luna event sequence through `normaliz | Defense | `src/toolCallAccumulator.ts` | An id is captured **once** (first fragment that carries one) and never overwritten by later fragments | | Self-heal | `src/responsesRequest.ts` | `pairResponsesFunctionCallItems` collapses duplicate `function_call` items sharing one call_id (keeps the first) — histories already poisoned by 0.7.7 recover automatically | +### Identity Rule extended to every Responses entry point + +The post-fix audit found the same latent fallback in two more places and removed it — **no Responses path may ever derive call identity from `item.id`**: + +- `output_item.added` handler: `firstString(item.call_id, item.id)` → `call_id` only (empty → the flush fabricates a unique id, which cannot collide). +- `normalizeResponsesFullResponse` function_call mapping: same fallback removed. + +Gateways that always send `call_id` (all known real shapes) are unaffected; only the hypothetical call_id-less shape changes behavior, and strictly for the better. + The `argumentsDone` REPLACE repair for arguments (the actual #244 fix) is preserved and pinned by test. ## Tests -`src/test/issue244-followup-regression.test.ts` (renamed from the repro): +`src/test/issue244-followup-regression.test.ts` (renamed from the repro) — 6 tests: 1. Part id stays `call_*` when done carries only `item_id` (the exact reported shape). 2. Full round-trip: wire request pairs function_call with its output. 3. `argumentsDone` still repairs mis-joined arguments (#244 fix preserved). 4. Pairing collapses duplicate function_call items (poisoned-history self-heal). +5. `output_item.added` without `call_id` never adopts the `fc_` item id. +6. `normalizeResponsesFullResponse` maps `call_id` only (no `item.id` fallback). + +### Pre-release E2E (`tmp/e2e-issue244-prerelease.mjs` — 17/17 checks) + +Full Copilot Chat loop simulated without VS Code, against the captured luna shapes: + +- Turn 1: stream → part id = gateway `call_id`, arguments `"what is 2 plus 2?"` keep their spaces (the original #244 symptom). +- Turn 2: history replay → wire request → pairing call/output = 1:1, no 400 possible. +- Turn 3: second turn reusing item id `fc_1` (the 0.7.7 poison) stays clean. +- Poisoned-history self-heal: a history already written by 0.7.7 collapses to 1 call + 1 output — users recover without clearing the chat. +- Non-stream full-response path sanity. -Verification: **493/493** tests pass, `npm run compile` clean. +Verification: **495/495** unit tests, `npm run lint` 7/7, retry E2E mock server 9/9, `npm run compile` clean. ## Lesson Recorded diff --git a/src/core/routing.ts b/src/core/routing.ts index 6677e6f42..c18e0b818 100644 --- a/src/core/routing.ts +++ b/src/core/routing.ts @@ -78,6 +78,10 @@ export function normalizeResponsesStreamEvent(data: unknown): unknown { if (eventType === "response.output_item.added") { const item = data.item; if (isRecord(item) && item.type === "function_call" && typeof item.name === "string") { + // IDENTITY RULE (#244 follow-up): call identity comes ONLY from + // `call_id`. Never fall back to `item.id` (`fc_*`) — item ids are + // reused across turns and would collide in replayed history. + const callId = typeof item.call_id === "string" && item.call_id.trim() ? item.call_id : ""; return { choices: [ { @@ -86,7 +90,7 @@ export function normalizeResponsesStreamEvent(data: unknown): unknown { tool_calls: [ { index: typeof data.output_index === "number" ? data.output_index : 0, - id: firstString(item.call_id, item.id) ?? "", + id: callId, type: "function", function: { name: item.name, arguments: "" }, }, @@ -242,7 +246,9 @@ export function normalizeResponsesFullResponse(data: unknown): unknown { if (item.type === "function_call" && typeof item.name === "string") { toolCalls.push({ - id: firstString(item.call_id, item.id) ?? "", + // IDENTITY RULE (#244 follow-up): call_id only — no item.id fallback + // (fc_* item ids are reused across turns; see added-handler above). + id: firstString(item.call_id) ?? "", type: "function", function: { name: item.name, diff --git a/src/test/issue244-followup-regression.test.ts b/src/test/issue244-followup-regression.test.ts index 43ce1c647..d646b299b 100644 --- a/src/test/issue244-followup-regression.test.ts +++ b/src/test/issue244-followup-regression.test.ts @@ -168,3 +168,36 @@ describe("#244 follow-up: done event repairs arguments, never identity", () => { assert.equal(outs.length, 1, `expected 1 function_call_output after dedupe, got ${String(outs.length)}`); }); }); + +describe("#244 follow-up: identity rule — no item.id fallback anywhere", () => { + before(async () => { + const routing = await import("../core/routing.js"); + normalizeResponsesStreamEvent = routing.normalizeResponsesStreamEvent; + const responses = await import("../responsesRequest.js"); + responsesInputItemsFromMessage = responses.responsesInputItemsFromMessage; + pairResponsesFunctionCallItems = responses.pairResponsesFunctionCallItems; + }); + + it("output_item.added without call_id does NOT adopt the fc_ item id", () => { + const result = normalizeResponsesStreamEvent({ + type: "response.output_item.added", + output_index: 0, + // No call_id — the old firstString(item.call_id, item.id) fallback + // adopted "fc_1", which is reused across turns. + item: { id: "fc_1", type: "function_call", status: "in_progress", name: "read_file", arguments: "" }, + }) as { choices: { delta: { tool_calls: { id?: string }[] } }[] }; + const call = result.choices[0].delta.tool_calls[0]; + assert.notEqual(call.id, "fc_1", "item id must never become call identity"); + }); + + it("normalizeResponsesFullResponse maps function_call call_id only (no item.id fallback)", async () => { + const routing = await import("../core/routing.js"); + const full = routing.normalizeResponsesFullResponse({ + response: { + output: [{ type: "function_call", id: "fc_7", call_id: "call_OK", name: "search", arguments: '{"q":"x"}' }], + stop_reason: "tool_calls", + }, + }) as { choices: { message: { tool_calls: { id: string }[] } }[] }; + assert.equal(full.choices[0]?.message.tool_calls[0]?.id, "call_OK"); + }); +});