From ce909b4bce3f1995675e9be7086c7dd1cff920a4 Mon Sep 17 00:00:00 2001 From: mdwsk88 <924038395@qq.com> Date: Sat, 26 Sep 2026 18:52:16 +0800 Subject: [PATCH 1/4] fix(server): admit direct mcp__ tool calls through code-mode exec In desktop code mode the host declares only `exec`, yet models occasionally address MCP tools directly as `mcp__server__tool`. The undeclared-tool guard (#1700) fails closed and the routed turn dies with a 502, surfacing in the Codex desktop app as a "reconnecting N/5" retry banner; the retry usually succeeds because the model then wraps the same call in exec. Observed with Kimi K3 and GLM 5.3 routed through codebuddy-cn and AI2API. Normalize direct `mcp__*` names to the exec helper in normalizeDeclaredToolName (subject to the same legacy-bridge exclusion gate as other helper rewrites) and compile them to an explicit `await tools.(...)` call in compileCodeModeHelperInput, with bracket access for non-identifier names. The call now reaches the host MCP surface instead of being rejected; unknown names still fail safe on the host side. Also re-export the new predicate from the types barrel. --- src/responses/code-mode-helper-compat.ts | 15 ++- src/types.ts | 1 + src/types/tools.ts | 37 +++++-- .../bridge-legacy-shell-normalization.test.ts | 33 +++++++ .../responses-code-mode-mcp-direct.test.ts | 99 +++++++++++++++++++ 5 files changed, 177 insertions(+), 8 deletions(-) create mode 100644 tests/responses/responses-code-mode-mcp-direct.test.ts diff --git a/src/responses/code-mode-helper-compat.ts b/src/responses/code-mode-helper-compat.ts index 4d4721288ac..6fad7be0c3d 100644 --- a/src/responses/code-mode-helper-compat.ts +++ b/src/responses/code-mode-helper-compat.ts @@ -3,13 +3,16 @@ import { normalizeApplyPatchDelimiters, unwrapFreeformToolInput, } from "./apply-patch-envelope"; -import { declaresCodeModeExec } from "../types/tools"; +import { declaresCodeModeExec, isCodeModeMcpDirectName } from "../types/tools"; import { parseCodeModeShellInput } from "./code-mode-shell-input"; function isPlainObject(value: unknown): value is Record { return !!value && typeof value === "object" && !Array.isArray(value); } +/** A nested host tool reachable as `tools.` when the flattened name is one identifier. */ +const CODE_MODE_IDENTIFIER_NAME = /^[A-Za-z_$][A-Za-z0-9_$]*$/; + /** * Convert a nested Code Mode helper call into unified-exec JavaScript. * @@ -95,6 +98,16 @@ export function compileCodeModeHelperInput( if (helperName === "create_goal" || helperName === "get_goal" || helperName === "update_goal") { return `const result = await tools.${helperName}(${JSON.stringify(args)});\ntext(result);`; } + if (isCodeModeMcpDirectName(helperName)) { + // Direct `mcp____` call under a code-mode catalog: the emitted name IS the + // nested host tool's name, so compile to the same `tools.(args)` the model could + // have written. Dot access when the name is a clean identifier (the common case); bracket + // access otherwise, so a hyphenated server name still addresses the same tool. + const target = CODE_MODE_IDENTIFIER_NAME.test(helperName) + ? `tools.${helperName}` + : `tools[${JSON.stringify(helperName)}]`; + return `const result = await ${target}(${JSON.stringify(args)});\ntext(result);`; + } return `const result = await tools.exec_command(${JSON.stringify(args)});\ntext(result);`; } diff --git a/src/types.ts b/src/types.ts index ff961d813e4..d380442c649 100644 --- a/src/types.ts +++ b/src/types.ts @@ -8,6 +8,7 @@ export { dottedToolName, namespacedToolName, normalizeDeclaredToolName, + isCodeModeMcpDirectName, toolChoiceAliases, createToolChoiceResolver, toolChoiceCandidates, diff --git a/src/types/tools.ts b/src/types/tools.ts index c9066d01b58..76b8b3c7eaa 100644 --- a/src/types/tools.ts +++ b/src/types/tools.ts @@ -70,8 +70,9 @@ export function dottedToolName(namespace: string | undefined, name: string): str * `apply_patch`, `view_image`, or one of the goal helpers (`create_goal`, `get_goal`, * `update_goal`, #5495) instead of the declared `exec`. Accept these nested helper names only * when the request catalog actually declares `exec` and does not itself declare the emitted name - * (an MCP server may legitimately advertise one under its own namespace). The list is closed: an - * unlisted name is never admitted through `exec`. + * (an MCP server may legitimately advertise one under its own namespace). The helper list itself + * is closed; the one other admission through `exec` is a direct `mcp____` call, + * which `isCodeModeMcpDirectName` recognizes. */ const LEGACY_SHELL_BRIDGE_TOOL_NAMES = ["exec_command", "shell_command"] as const; const CODE_MODE_HELPER_TOOL_NAMES = [ @@ -105,6 +106,23 @@ export const CODE_MODE_HELPER_WIRE_NAMES: ReadonlySet = new Set( CODE_MODE_HELPER_TOOL_NAMES, ); +/** + * A flattened MCP wire name (`mcp____`) emitted as a direct tool call. + * + * Codex code mode reaches the host's nested tools through `tools.(...)` inside `exec`, + * so none of them are declared; routed models (observed: Kimi K3, GLM 5.3) occasionally skip the + * wrapper and call the flattened name directly. Under a code-mode catalog the call is compiled + * into the equivalent `tools.(...)` exec body instead of failing closed — capability- + * equivalent, since the model could have written that JavaScript itself. Server and tool must + * both be non-empty so a bare `mcp__` prefix never qualifies. + */ +export function isCodeModeMcpDirectName(name: string): boolean { + if (!name.startsWith("mcp__")) return false; + const rest = name.slice("mcp__".length); + const separator = rest.indexOf("__"); + return separator > 0 && separator + "__".length < rest.length; +} + /** * Spellings that may never be MANUFACTURED as a bare alias for a namespaced tool. * @@ -137,8 +155,9 @@ export const NAMESPACED_BARE_ALIAS_EXCLUDED_NAMES: ReadonlySet = new Set * The same wrapper may surround an already-flattened namespace identity; accept that exact * declared suffix without treating its child name as a bare declaration. * Also normalizes nested helper names (`exec_command`, `shell_command`, `write_stdin`, - * `apply_patch`, `view_image`, `create_goal`, `get_goal`, `update_goal`) to - * `exec` when code-mode `exec` is declared in the request catalog. + * `apply_patch`, `view_image`, `create_goal`, `get_goal`, `update_goal`) and direct + * `mcp____` calls to `exec` when code-mode `exec` is declared in the + * request catalog. * * @param name - The tool name emitted on the wire by the provider. * @param declared - All wire tool names declared in the request catalog, including aliases. @@ -192,9 +211,13 @@ export function normalizeDeclaredToolName( if ((LEGACY_SHELL_BRIDGE_TOOL_NAMES as readonly string[]).some(legacy => declared.has(legacy))) { return candidate; } - return (CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(candidate) - ? CODE_MODE_EXEC_TOOL_NAME - : candidate; + if ((CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(candidate)) { + return CODE_MODE_EXEC_TOOL_NAME; + } + // A direct `mcp____` call names a nested host tool the code-mode catalog + // never declares; `compileCodeModeHelperInput` turns it into the `tools.(...)` + // exec body the model could have written itself. + return isCodeModeMcpDirectName(candidate) ? CODE_MODE_EXEC_TOOL_NAME : candidate; } /** diff --git a/tests/adapters/bridge-legacy-shell-normalization.test.ts b/tests/adapters/bridge-legacy-shell-normalization.test.ts index cc16fdec090..b1d1df5cb28 100644 --- a/tests/adapters/bridge-legacy-shell-normalization.test.ts +++ b/tests/adapters/bridge-legacy-shell-normalization.test.ts @@ -160,4 +160,37 @@ describe("bridge normalizes code-mode helper names against the declared catalog" expect(sse).not.toContain("tools.view_image"); expect(sse).not.toContain('"name":"exec"'); }); + + // Codex Desktop code mode declares only `exec`; routed models (observed: Kimi K3, GLM 5.3) + // sometimes call a nested host tool by its flattened `mcp____` name directly. + // The guard failed those turns closed, which the desktop surfaced as reconnect banners. + test("a direct mcp tool call is delivered as exec running the nested host call", async () => { + const sse = await drain(bridgeToResponsesSSE( + toolTurn("mcp__codex_app__get_usage_limits", "{}"), + "fixture-model", + undefined, + new Set(["exec"]), + undefined, + undefined, + 50_000, + { declaredToolNames: new Set(["exec"]) }, + )); + expect(sse).not.toContain("undeclared client tool"); + expect(sse).toContain('"name":"exec"'); + expect(sse).toContain('await tools.mcp__codex_app__get_usage_limits({})'); + }); + + test("a malformed mcp-prefixed name still fails the turn", async () => { + const sse = await drain(bridgeToResponsesSSE( + toolTurn("mcp__solo", "{}"), + "fixture-model", + undefined, + new Set(["exec"]), + undefined, + undefined, + 50_000, + { declaredToolNames: new Set(["exec"]) }, + )); + expect(sse).toContain("undeclared client tool"); + }); }); diff --git a/tests/responses/responses-code-mode-mcp-direct.test.ts b/tests/responses/responses-code-mode-mcp-direct.test.ts new file mode 100644 index 00000000000..8179a42c437 --- /dev/null +++ b/tests/responses/responses-code-mode-mcp-direct.test.ts @@ -0,0 +1,99 @@ +import { describe, expect, test } from "bun:test"; +import { restoreRoutedCustomCallsInJson } from "../../src/responses/custom-tool-compat"; +import { compileCodeModeHelperInput } from "../../src/responses/code-mode-helper-compat"; +import { undeclaredToolCallNameInResponse } from "../../src/server/responses-undeclared-tool-guard"; +import { isCodeModeMcpDirectName, normalizeDeclaredToolName } from "../../src/types/tools"; + +const CODE_MODE = new Set(["exec"]); + +// Codex Desktop code mode declares only the freeform `exec` shell; every nested host tool +// (`tools.mcp__codex_app__get_usage_limits`, ...) is reachable through it but undeclared. +// Routed models (observed: Kimi K3, GLM 5.3) sometimes call the flattened MCP name directly +// instead of wrapping it in exec JavaScript, and the undeclared-tool guard failed those turns +// closed — visible in the desktop client as "reconnecting N/5" banners. These pin the +// normalization that compiles such a call into the exec body the model could have written. +describe("code-mode direct mcp tool-call recovery", () => { + test("recognizes only well-formed flattened mcp names", () => { + expect(isCodeModeMcpDirectName("mcp__codex_app__get_usage_limits")).toBe(true); + expect(isCodeModeMcpDirectName("mcp__my-server__do_thing")).toBe(true); + for (const name of ["mcp__", "mcp__server", "mcp____tool", "mcp__server__", "exec_command", "mcp_x__y"]) { + expect(isCodeModeMcpDirectName(name)).toBe(false); + } + }); + + test("maps a direct mcp call through a declared exec only", () => { + expect(normalizeDeclaredToolName("mcp__codex_app__get_usage_limits", CODE_MODE)).toBe("exec"); + expect(normalizeDeclaredToolName("mcp__codex_app__get_usage_limits", new Set())).toBe("mcp__codex_app__get_usage_limits"); + // A catalog that declares the name itself keeps the call's own identity. + expect(normalizeDeclaredToolName( + "mcp__codex_app__get_usage_limits", + new Set(["exec", "mcp__codex_app__get_usage_limits"]), + )).toBe("mcp__codex_app__get_usage_limits"); + // The flat-bridge shape (legacy shell names declared next to exec) is not code mode. + expect(normalizeDeclaredToolName( + "mcp__codex_app__get_usage_limits", + new Set(["exec", "exec_command"]), + )).toBe("mcp__codex_app__get_usage_limits"); + // Malformed mcp-ish names stay undeclared. + expect(normalizeDeclaredToolName("mcp__server", CODE_MODE)).toBe("mcp__server"); + }); + + test("compiles the call to the matching nested host tool", () => { + expect(compileCodeModeHelperInput('{"limit":3}', "mcp__codex_app__list_threads")).toBe( + 'const result = await tools.mcp__codex_app__list_threads({"limit":3});\ntext(result);', + ); + // A hyphenated name is not one identifier: bracket access addresses the same tool. + expect(compileCodeModeHelperInput("{}", "mcp__my-server__do_thing")).toBe( + 'const result = await tools["mcp__my-server__do_thing"]({});\ntext(result);', + ); + // Malformed provider text stays data; nested-tool validation rejects it, not JavaScript. + expect(compileCodeModeHelperInput("not json", "mcp__x__y")).toBe( + 'const result = await tools.mcp__x__y("not json");\ntext(result);', + ); + }); + + test("keeps the undeclared-tool guard admit/block boundary unchanged elsewhere", () => { + const source = { + output: [{ + type: "function_call", + id: "fc_mcp", + call_id: "call_mcp", + name: "mcp__codex_app__get_usage_limits", + arguments: "{}", + }], + }; + expect(undeclaredToolCallNameInResponse(source, CODE_MODE)).toBeUndefined(); + expect(undeclaredToolCallNameInResponse(source, new Set())).toBe("mcp__codex_app__get_usage_limits"); + // A name that only looks mcp-ish is still blocked under a code-mode catalog. + const malformed = { + output: [{ type: "function_call", id: "fc_x", call_id: "call_x", name: "mcp__solo", arguments: "{}" }], + }; + expect(undeclaredToolCallNameInResponse(malformed, CODE_MODE)).toBe("mcp__solo"); + }); + + test("restores a recorded direct mcp call as the declared exec", () => { + const source = { + output: [{ + type: "function_call", + id: "fc_mcp", + call_id: "call_mcp", + name: "mcp__codex_app__get_usage_limits", + arguments: "{}", + }], + }; + const restored = JSON.parse(restoreRoutedCustomCallsInJson( + JSON.stringify(source), + CODE_MODE, + new Set(), + CODE_MODE, + )); + expect(restored.output).toMatchObject([{ + type: "custom_tool_call", + name: "exec", + call_id: "call_mcp", + input: 'const result = await tools.mcp__codex_app__get_usage_limits({});\ntext(result);', + }]); + expect(undeclaredToolCallNameInResponse(restored, CODE_MODE)).toBeUndefined(); + }); +}); + From 89236a58dee2f4bf795a23b6b350483a9af886d3 Mon Sep 17 00:00:00 2001 From: mdwsk88 <924038395@qq.com> Date: Sun, 27 Sep 2026 08:19:37 +0800 Subject: [PATCH 2/4] fix(responses): require custom exec provenance for direct MCP recovery Direct mcp__server__tool calls in code mode are recovered only when the request explicitly declares a custom exec tool. The undeclared-tool guard now derives bare custom declarations and grants recovery only on delivery paths that actually restore converted custom calls, so an ordinary JSON function named exec cannot be mistaken for a JavaScript executor. --- .../content/docs/guides/codex-integration.md | 9 + scripts/test-layout/layout.json | 1 + src/bridge/response-json.ts | 4 +- src/bridge/sse.ts | 4 +- src/responses/custom-tool-compat.ts | 2 +- src/server/responses-undeclared-tool-guard.ts | 49 ++++- src/server/responses/adapter-delivery.ts | 8 +- src/server/responses/passthrough-delivery.ts | 4 + src/server/responses/passthrough-dispatch.ts | 10 + src/server/responses/run-turn-execution.ts | 6 +- src/types/tools.ts | 17 +- structure/providers-and-adapters.md | 4 + structure/transports/responses-wire-shapes.md | 22 ++- .../bridge-legacy-shell-normalization.test.ts | 4 +- tests/fixtures/test-layout-expected.json | 1 + .../responses-code-mode-mcp-direct.test.ts | 185 +++++++++++++++--- 16 files changed, 276 insertions(+), 54 deletions(-) diff --git a/docs-site/src/content/docs/guides/codex-integration.md b/docs-site/src/content/docs/guides/codex-integration.md index 1b964b14529..2089cd136e9 100644 --- a/docs-site/src/content/docs/guides/codex-integration.md +++ b/docs-site/src/content/docs/guides/codex-integration.md @@ -653,6 +653,15 @@ code-mode `exec` has the call converted into the matching `tools.(...)` `exec`. A catalog that genuinely declares the bare goal tool keeps it, and a catalog that declares neither the tool nor `exec` still rejects the call as undeclared. +On routed conversions with a verified freeform code-mode `exec` catalog, structured calls +sent directly to +`mcp____` (including a provider-added `default.` prefix) are also +wrapped as nested host-tool calls. This avoids a retry +caused solely by a model omitting the `exec` wrapper. Explicitly declared MCP tools +keep their normal behavior; an ordinary JSON function named `exec` does not enable +this repair. Unknown tools still fail at the host. Tool-call records printed as +ordinary answer text are not executed by this compatibility rule. + For routed Responses turns, an explicit tool-enforcement policy also rejects client tool calls if the request's declared-tool catalog is unavailable. An empty declared catalog rejects every client tool call; Chat and Anthropic clients retain their own tool-validation responsibility. diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index e8c6e95aa0d..082cd6992e6 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -1549,6 +1549,7 @@ "responses-bare-echo-helper-fence.test.ts": "responses", "responses-canonical-only-top-level-fields.test.ts": "responses", "responses-code-mode-goal-helpers.test.ts": "responses", + "responses-code-mode-mcp-direct.test.ts": "responses", "responses-code-mode-patch-compile.test.ts": "responses", "responses-code-mode-shell-compile.test.ts": "responses", "responses-compact-handoff-admission.test.ts": "responses", diff --git a/src/bridge/response-json.ts b/src/bridge/response-json.ts index 48fef166e08..c3f57d6ca01 100644 --- a/src/bridge/response-json.ts +++ b/src/bridge/response-json.ts @@ -86,6 +86,8 @@ function buildResponseJSONWithBudget( toolNsMap?: Map; /** Request-visible tool names. Required for client calls when enforcement is explicitly enabled. */ declaredToolNames?: ReadonlySet; + /** Bare custom declarations; unlike freeformToolNames, excludes foreign namespace children. */ + bareCustomToolNames?: ReadonlySet; /** See `bridgeToResponsesSSE`: enforcement is separate from normalization (#4735). */ enforceDeclaredToolNames?: boolean; /** Declared parameter schema per tool name; repairs integral-float integer args (#1611). */ @@ -451,7 +453,7 @@ function buildResponseJSONWithBudget( rememberReasoningForCall(e.id, rawReasoningForNextToolCall, replayCacheScope); } flushToolCall(); - const effectiveName = normalizeDeclaredToolName(e.name, options?.declaredToolNames); + const effectiveName = normalizeDeclaredToolName(e.name, options?.declaredToolNames, undefined, options?.bareCustomToolNames); if ( (options?.enforceDeclaredToolNames === true || options?.declaredToolNames != null) && options?.enforceDeclaredToolNames !== false diff --git a/src/bridge/sse.ts b/src/bridge/sse.ts index c733c1e45d7..730d67dacac 100644 --- a/src/bridge/sse.ts +++ b/src/bridge/sse.ts @@ -101,6 +101,8 @@ export function bridgeToResponsesSSE( onUsage?: (usage: OcxUsage | undefined) => void; /** Request-visible tool names. Required for client calls when enforcement is explicitly enabled. */ declaredToolNames?: ReadonlySet; + /** Bare custom declarations; unlike freeformToolNames, excludes foreign namespace children. */ + bareCustomToolNames?: ReadonlySet; /** * Whether `declaredToolNames` is an authorization boundary this proxy enforces, or only the * catalog used to normalize provider-invented names back to declared ones. @@ -1013,7 +1015,7 @@ export function bridgeToResponsesSSE( rememberReasoningForCall(event.id, rawReasoningForNextToolCall, replayCacheScope); } if (currentToolCall) closeCurrentToolCall(); - const effectiveName = normalizeDeclaredToolName(event.name, options?.declaredToolNames); + const effectiveName = normalizeDeclaredToolName(event.name, options?.declaredToolNames, undefined, options?.bareCustomToolNames); const codeModeHelperName = effectiveName === "exec" && event.name !== effectiveName ? event.name : undefined; diff --git a/src/responses/custom-tool-compat.ts b/src/responses/custom-tool-compat.ts index 989af132974..4b15bc9f9ff 100644 --- a/src/responses/custom-tool-compat.ts +++ b/src/responses/custom-tool-compat.ts @@ -76,7 +76,7 @@ export function routedCustomToolTargetName( if (wireName === undefined) return undefined; if (names.has(wireName)) return wireName; if (!isPlainObject(value) || typeof value.namespace === "string") return undefined; - const normalized = normalizeDeclaredToolName(wireName, declaredNames); + const normalized = normalizeDeclaredToolName(wireName, declaredNames, undefined, names); return normalized !== wireName && names.has(normalized) ? normalized : undefined; } diff --git a/src/server/responses-undeclared-tool-guard.ts b/src/server/responses-undeclared-tool-guard.ts index 943f19e7d6d..52cc338e509 100644 --- a/src/server/responses-undeclared-tool-guard.ts +++ b/src/server/responses-undeclared-tool-guard.ts @@ -226,6 +226,35 @@ export function collectDeclaredBareWireToolNames(body: unknown): Set { return names; } +/** Custom declarations with no foreign namespace, for code-mode exec recovery. */ +export function collectDeclaredBareCustomWireToolNames(body: unknown): Set { + const names = new Set(); + if (!isPlainObject(body)) return names; + const specGroups: unknown[] = [body.tools]; + if (Array.isArray(body.input)) { + for (const item of body.input) { + if (isPlainObject(item) && (item.type === "additional_tools" || item.type === "tool_search_output")) { + specGroups.push(item.tools); + } + } + } + const add = (tool: unknown): void => { + if (isPlainObject(tool) && tool.type === "custom" && typeof tool.name === "string") names.add(tool.name); + }; + for (const specs of specGroups) { + if (!Array.isArray(specs)) continue; + for (const spec of specs) { + if (!isPlainObject(spec)) continue; + if (spec.type === "namespace") { + if (spec.name === BUILTIN_FUNCTIONS_NAMESPACE && Array.isArray(spec.tools)) { + for (const inner of spec.tools) add(inner); + } + } else add(spec); + } + } + return names; +} + function addNamelessClientCallTypes(callTypes: Set, specs: unknown): void { if (!Array.isArray(specs)) return; for (const spec of specs) { @@ -349,6 +378,7 @@ export function hasExplicitWireToolCatalog(body: unknown): boolean { * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns The undeclared tool call name if unauthorized, or undefined if permitted. */ function undeclaredNameInItem( @@ -357,6 +387,7 @@ function undeclaredNameInItem( declaredNamelessClientCallTypes: ReadonlySet, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string | undefined { if (!isPlainObject(item)) return undefined; if (typeof item.type !== "string") return undefined; @@ -396,7 +427,7 @@ function undeclaredNameInItem( ) return undefined; return name; } - const effectiveName = normalizeDeclaredToolName(name, declared, declaredBare); + const effectiveName = normalizeDeclaredToolName(name, declared, declaredBare, declaredCustom); if (declared.has(effectiveName)) return undefined; return name; } @@ -409,6 +440,7 @@ function undeclaredNameInItem( * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns The name of the first undeclared tool call, or undefined. */ export function undeclaredToolCallName( @@ -417,18 +449,19 @@ export function undeclaredToolCallName( declaredNamelessClientCallTypes: ReadonlySet = EMPTY_DECLARED_NAMELESS_CLIENT_CALL_TYPES, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string | undefined { if (!isPlainObject(payload)) return undefined; if (payload.type === "response.output_item.added" || payload.type === "response.output_item.done") { - return undeclaredNameInItem(payload.item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + return undeclaredNameInItem(payload.item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); } if (payload.type === "response.function_call_arguments.done" && typeof payload.name === "string") { const fakeItem = { type: "function_call", name: payload.name, namespace: payload.namespace }; - return undeclaredNameInItem(fakeItem, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + return undeclaredNameInItem(fakeItem, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); } // Sparse gateways skip incremental items and only ever ship the terminal snapshot. if (payload.type === "response.completed" || payload.type === "response.incomplete") { - return undeclaredToolCallNameInResponse(payload.response, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + return undeclaredToolCallNameInResponse(payload.response, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); } return undefined; } @@ -441,6 +474,7 @@ export function undeclaredToolCallName( * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns The name of the first undeclared tool call, or undefined. */ export function undeclaredToolCallNameInResponse( @@ -449,10 +483,11 @@ export function undeclaredToolCallNameInResponse( declaredNamelessClientCallTypes: ReadonlySet = EMPTY_DECLARED_NAMELESS_CLIENT_CALL_TYPES, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string | undefined { if (!isPlainObject(response) || !Array.isArray(response.output)) return undefined; for (const item of response.output) { - const name = undeclaredNameInItem(item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + const name = undeclaredNameInItem(item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); if (name !== undefined) return name; } return undefined; @@ -667,6 +702,7 @@ function failedBlocks(name: string, newline: string): readonly string[] { * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns An SSE block rewrite function. */ export function createUndeclaredToolCallGuardBlockRewrite( @@ -674,6 +710,7 @@ export function createUndeclaredToolCallGuardBlockRewrite( declaredNamelessClientCallTypes: ReadonlySet = EMPTY_DECLARED_NAMELESS_CLIENT_CALL_TYPES, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): SseBlockRewrite { let tripped = false; return (block: string) => { @@ -686,7 +723,7 @@ export function createUndeclaredToolCallGuardBlockRewrite( } catch { return [block]; } - const name = undeclaredToolCallName(parsed, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + const name = undeclaredToolCallName(parsed, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); if (name !== undefined) { tripped = true; return failedBlocks(name, block.includes("\r\n") ? "\r\n" : "\n"); diff --git a/src/server/responses/adapter-delivery.ts b/src/server/responses/adapter-delivery.ts index 58a445b88e7..f327e0137a3 100644 --- a/src/server/responses/adapter-delivery.ts +++ b/src/server/responses/adapter-delivery.ts @@ -119,7 +119,7 @@ export async function deliverAdapterResponse( continuation: fetchGuardedEmptyCompletionRetry, }) : eventStream; - const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames } = toolBridgeMaps; + const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames } = toolBridgeMaps; // One completion owner for both deliveries: the bridge calls it from its terminal, the // direct client encoder from the fold of the same events. const onCompletedResponse = (response: Record, providerState?: OcxProviderContinuationState) => { @@ -176,8 +176,9 @@ export async function deliverAdapterResponse( localUpstream, hideThinkingSummary: parsed.options.hideThinkingSummary, declaredToolNames, + bareCustomToolNames, enforceDeclaredToolNames: options.inboundWire !== "chat" && options.inboundWire !== "anthropic", - toolParameterSchemas, + toolParameterSchemas, ...(options.onFirstOutput ? { onFirstOutput: options.onFirstOutput } : {}), ...(routedCompaction ? { compaction: true } : {}), // Same grok-surface split as the runTurn branch above. @@ -236,7 +237,7 @@ export async function deliverAdapterResponse( } finally { cleanupUpstreamAbort(); } - const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames } = toolBridgeMaps; + const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames } = toolBridgeMaps; let providerState: OcxProviderContinuationState | undefined; const json = buildResponseJSON(events, parsed._responseModelId ?? parsed.modelId, { translatorBudget, @@ -244,6 +245,7 @@ export async function deliverAdapterResponse( hideThinkingSummary: parsed.options.hideThinkingSummary, toolNsMap, declaredToolNames, + bareCustomToolNames, enforceDeclaredToolNames: options.inboundWire !== "chat" && options.inboundWire !== "anthropic", toolParameterSchemas, freeformToolNames, diff --git a/src/server/responses/passthrough-delivery.ts b/src/server/responses/passthrough-delivery.ts index 366c6c00ddc..4c039f0d55c 100644 --- a/src/server/responses/passthrough-delivery.ts +++ b/src/server/responses/passthrough-delivery.ts @@ -313,6 +313,7 @@ export async function deliverPassthroughResponse( | "declaredNamelessClientCallTypes" | "providerExecutedCallTypes" | "declaredBareWireToolNames" + | "recoverableBareCustomWireToolNames" | "rememberPassthroughResponse" | "noteInspectedPayload" | "normalizeFunctionCompletionJson" @@ -336,6 +337,7 @@ export async function deliverPassthroughResponse( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, rememberPassthroughResponse, noteInspectedPayload, normalizeFunctionCompletionJson, @@ -732,6 +734,7 @@ export async function deliverPassthroughResponse( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ) : undefined, grokUpstreamEchoEnabled @@ -998,6 +1001,7 @@ export async function deliverPassthroughResponse( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ); } catch { return undefined; diff --git a/src/server/responses/passthrough-dispatch.ts b/src/server/responses/passthrough-dispatch.ts index 1b953bc0f8f..b78be9e0a97 100644 --- a/src/server/responses/passthrough-dispatch.ts +++ b/src/server/responses/passthrough-dispatch.ts @@ -22,6 +22,7 @@ import { hasExplicitWireToolCatalog, collectDeclaredWireToolNames, collectDeclaredBareWireToolNames, + collectDeclaredBareCustomWireToolNames, collectDeclaredNamelessClientCallTypes, collectProviderExecutedCallTypes, undeclaredToolCallName, @@ -314,6 +315,7 @@ export async function preparePassthroughExchange( const clientExplicitWireToolCatalog = hasExplicitWireToolCatalog(clientToolAuthorizationBody); const clientDeclaredWireToolNames = collectDeclaredWireToolNames(clientToolAuthorizationBody); const clientDeclaredBareWireToolNames = collectDeclaredBareWireToolNames(clientToolAuthorizationBody); + const clientDeclaredBareCustomWireToolNames = collectDeclaredBareCustomWireToolNames(clientToolAuthorizationBody); const clientDeclaredNamelessCallTypes = collectDeclaredNamelessClientCallTypes( clientToolAuthorizationBody, ); @@ -361,6 +363,11 @@ export async function preparePassthroughExchange( ) routedCustomToolRepairNames.add(name); } } + // Only grant direct MCP recovery when this delivery actually restores converted custom + // calls. Native forward and injection paths have no such rewrite. + const recoverableBareCustomWireToolNames = new Set( + [...clientDeclaredBareCustomWireToolNames].filter(name => routedCustomToolNames.has(name)), + ); for (const name of request.convertedRoutedToolSearchNames ?? []) { // The adapter already keeps this set empty when tool_choice forbids the private search. // Its wire name may be collision-aliased, so comparing it to the caller-facing name here @@ -560,6 +567,7 @@ export async function preparePassthroughExchange( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ) !== undefined) { inspectionSawUndeclaredTool = true; } @@ -605,6 +613,7 @@ export async function preparePassthroughExchange( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ) !== undefined ) { return; @@ -1815,6 +1824,7 @@ export async function preparePassthroughExchange( }, declaredWireToolNames, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, declaredNamelessClientCallTypes, authorizedBareNamespaceToolAliases, normalizeFunctionCompletionJson, diff --git a/src/server/responses/run-turn-execution.ts b/src/server/responses/run-turn-execution.ts index de0dc848117..736be252156 100644 --- a/src/server/responses/run-turn-execution.ts +++ b/src/server/responses/run-turn-execution.ts @@ -551,7 +551,7 @@ export async function executeResponsesRunTurn( return retryQueue.stream(); }; - const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames } = toolBridgeMaps; + const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames } = toolBridgeMaps; const enforceDeclaredToolNames = inboundWire !== "chat" && inboundWire !== "anthropic"; const classifyUndeclaredFirstTool = ( event: AdapterEvent, @@ -559,7 +559,7 @@ export async function executeResponsesRunTurn( if (!enforceDeclaredToolNames || event.type !== "tool_call_start") return undefined; // This tool is declared to the adapter by the private search loop. if (wsPlan && event.name === WEB_SEARCH_TOOL_NAME) return undefined; - const effectiveName = normalizeDeclaredToolName(event.name, declaredToolNames); + const effectiveName = normalizeDeclaredToolName(event.name, declaredToolNames, undefined, bareCustomToolNames); if (declaredToolNames.has(effectiveName)) return undefined; return { type: "error", @@ -673,6 +673,7 @@ export async function executeResponsesRunTurn( stallTimeoutSec, hideThinkingSummary: parsed.options.hideThinkingSummary, declaredToolNames, + bareCustomToolNames, enforceDeclaredToolNames, toolParameterSchemas, ...(options.onFirstOutput ? { onFirstOutput: options.onFirstOutput } : {}), @@ -815,6 +816,7 @@ export async function executeResponsesRunTurn( enforceDeclaredToolNames, toolParameterSchemas, freeformToolNames, + bareCustomToolNames, toolSearchToolNames, ...(routedCompaction ? { compaction: true } : {}), onProviderState: state => { providerState = state; }, diff --git a/src/types/tools.ts b/src/types/tools.ts index 76b8b3c7eaa..0c285ef86f1 100644 --- a/src/types/tools.ts +++ b/src/types/tools.ts @@ -157,18 +157,22 @@ export const NAMESPACED_BARE_ALIAS_EXCLUDED_NAMES: ReadonlySet = new Set * Also normalizes nested helper names (`exec_command`, `shell_command`, `write_stdin`, * `apply_patch`, `view_image`, `create_goal`, `get_goal`, `update_goal`) and direct * `mcp____` calls to `exec` when code-mode `exec` is declared in the - * request catalog. + * request catalog. MCP recovery additionally requires explicit custom-tool provenance; + * a structured function named `exec` is not a JavaScript executor. * * @param name - The tool name emitted on the wire by the provider. * @param declared - All wire tool names declared in the request catalog, including aliases. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. * When omitted, falls back to `declared`. + * @param declaredCustom - Custom wire identities from the caller's catalog, never bare aliases + * manufactured from foreign namespaces. * @returns The normalized tool name to expose downstream. */ export function normalizeDeclaredToolName( name: string, declared: ReadonlySet | undefined, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string { if (!declared) return name; if (declared.has(name)) return name; @@ -191,11 +195,12 @@ export function normalizeDeclaredToolName( candidate = bare; } else if ( // Code mode never declares bare helper names; a provider that invents `default.` - // for one still means the nested helper. Strip the prefix so the helper list - // below can rewrite it to `exec` (#4412). + // for one still means the nested helper. The same wrapper can surround a direct MCP + // name, but only a custom exec declaration authorizes that recovery. bare.length > 0 && declared.has(CODE_MODE_EXEC_TOOL_NAME) - && (CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(bare) + && ((CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(bare) + || (declaredCustom?.has(CODE_MODE_EXEC_TOOL_NAME) && isCodeModeMcpDirectName(bare))) && !declared.has("default." + bare) && !declared.has("default__" + bare) ) { @@ -217,7 +222,9 @@ export function normalizeDeclaredToolName( // A direct `mcp____` call names a nested host tool the code-mode catalog // never declares; `compileCodeModeHelperInput` turns it into the `tools.(...)` // exec body the model could have written itself. - return isCodeModeMcpDirectName(candidate) ? CODE_MODE_EXEC_TOOL_NAME : candidate; + return declaredCustom?.has(CODE_MODE_EXEC_TOOL_NAME) && isCodeModeMcpDirectName(candidate) + ? CODE_MODE_EXEC_TOOL_NAME + : candidate; } /** diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 511f0ce8d26..ad719ac37ec 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -13,6 +13,10 @@ the parser fails the turn before releasing their buffered calls. The capture-only bridge checks each raw tool-use start against the init handshake before buffering; a later init cannot authorize a call that started earlier. +Direct MCP names emitted in a verified custom code-mode catalog follow the +[Responses restoration boundary](transports/responses-wire-shapes.md#direct-mcp-calls-in-code-mode). +Ordinary structured functions named `exec` do not opt into this compatibility path. + RunTurn hosted search uses `src/web-search/run-turn-loop.ts`: synthetic calls remain private, progress reaches the bridge during collection, and a validated terminal precedes search execution. Complete search calls remain actionable at a truncated `done`; cancellation prevents subsequent queries and calls. OAuth preflight replay in `src/server/responses/run-turn-execution.ts` retains the synthetic tool while refreshing credential-scoped route state. In `src/server/responses/sidecar-execution.ts`, a search plan takes priority over image/video bridge execution for both transports; only fetch-capable adapters enter the fetch search loop. Combo preflight allows the private search tool only while a search plan is active; client tool declaration checks and replay-unsafe heartbeat protection remain enforced. diff --git a/structure/transports/responses-wire-shapes.md b/structure/transports/responses-wire-shapes.md index 1c05e1b578a..5205b0eea47 100644 --- a/structure/transports/responses-wire-shapes.md +++ b/structure/transports/responses-wire-shapes.md @@ -1,5 +1,19 @@ # Responses Wire Shapes +## Direct MCP calls in code mode + +On routed bridge or converted-custom passthrough paths, when the request declares a +freeform/custom code-mode `exec`, a structured call to +`mcp____` can be restored as an `exec` call to the matching nested host tool. +The same applies to a provider-added `default.` prefix when neither explicit `default.` nor +`default__` identity was declared. +The request must carry verified custom-tool provenance: an ordinary JSON function named +`exec` does not authorize this repair. Explicitly declared MCP tools keep their identity, +legacy shell catalogs stay unchanged, and unknown nested tools fail at the host. +Names and arguments are serialized as data; plain-text tool-call transcripts are never +promoted into executable calls by this rule. Native forwarding and injection lack this +restoration step, so their undeclared-tool guard still rejects a direct MCP call. + Per-wire request and stream shapes on the Responses data plane: mixed-wire model defaults, xAI agent-message continuation, declared-tool membership by inbound wire, and passthrough SSE stream shapes. The endpoint, dispatch, and credential rules they build on are in @@ -511,9 +525,11 @@ JavaScript. Ordinary JavaScript stays progressive. Coverage: `tests/responses/re An explicit custom-tool denial also requests recovery for unmapped historical results without a live catalog; history never adds current tool authorization. The custom-tool compatibility contract owns lowering and final validation. Muse may wrap an already-flattened namespace identity such as -`default.mcp__server__tool` only when the complete suffix exactly matches a declared namespaced name -and neither explicit `default.` nor `default__` identity exists. It cannot borrow a manufactured bare -alias; unknown suffixes still fail as undeclared tools. See [ADR-0099](../decisions/ADR-0099-responses-http-sse.md). +`default.mcp__server__tool` when the complete suffix exactly matches a declared namespaced name +and neither explicit `default.` nor `default__` identity exists. The custom code-mode `exec` +recovery above is a separate path for undeclared direct MCP names. Neither path can borrow a +manufactured bare alias. Outside code mode, unknown suffixes fail as undeclared tools; inside +code mode, the host rejects unknown nested tools. See [ADR-0099](../decisions/ADR-0099-responses-http-sse.md). > Decision record: [ADR-0099](../decisions/ADR-0099-responses-http-sse.md) diff --git a/tests/adapters/bridge-legacy-shell-normalization.test.ts b/tests/adapters/bridge-legacy-shell-normalization.test.ts index b1d1df5cb28..1530435dd50 100644 --- a/tests/adapters/bridge-legacy-shell-normalization.test.ts +++ b/tests/adapters/bridge-legacy-shell-normalization.test.ts @@ -173,7 +173,7 @@ describe("bridge normalizes code-mode helper names against the declared catalog" undefined, undefined, 50_000, - { declaredToolNames: new Set(["exec"]) }, + { declaredToolNames: new Set(["exec"]), bareCustomToolNames: new Set(["exec"]) }, )); expect(sse).not.toContain("undeclared client tool"); expect(sse).toContain('"name":"exec"'); @@ -189,7 +189,7 @@ describe("bridge normalizes code-mode helper names against the declared catalog" undefined, undefined, 50_000, - { declaredToolNames: new Set(["exec"]) }, + { declaredToolNames: new Set(["exec"]), bareCustomToolNames: new Set(["exec"]) }, )); expect(sse).toContain("undeclared client tool"); }); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 903440aa89d..f06874af27b 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1389,6 +1389,7 @@ "responses-bare-echo-helper-fence.test.ts": "responses", "responses-canonical-only-top-level-fields.test.ts": "responses", "responses-code-mode-goal-helpers.test.ts": "responses", + "responses-code-mode-mcp-direct.test.ts": "responses", "responses-code-mode-patch-compile.test.ts": "responses", "responses-code-mode-shell-compile.test.ts": "responses", "responses-compact-handoff-admission.test.ts": "responses", diff --git a/tests/responses/responses-code-mode-mcp-direct.test.ts b/tests/responses/responses-code-mode-mcp-direct.test.ts index 8179a42c437..d1d0cb34f87 100644 --- a/tests/responses/responses-code-mode-mcp-direct.test.ts +++ b/tests/responses/responses-code-mode-mcp-direct.test.ts @@ -1,10 +1,39 @@ import { describe, expect, test } from "bun:test"; -import { restoreRoutedCustomCallsInJson } from "../../src/responses/custom-tool-compat"; +import { bridgeToResponsesSSE, buildResponseJSON } from "../../src/bridge"; +import { restoreRoutedCustomCallsInJson, rewriteRoutedCustomToolsForUpstream } from "../../src/responses/custom-tool-compat"; import { compileCodeModeHelperInput } from "../../src/responses/code-mode-helper-compat"; -import { undeclaredToolCallNameInResponse } from "../../src/server/responses-undeclared-tool-guard"; +import { parseRequest } from "../../src/responses/parser"; +import { buildToolBridgeMaps } from "../../src/server/responses"; +import { createRoutedCustomToolRestoreBlockRewrite } from "../../src/server/responses-custom-tool-repair"; +import { + collectDeclaredBareCustomWireToolNames, + currentTurnWireToolCatalogBody, + undeclaredToolCallNameInResponse, +} from "../../src/server/responses-undeclared-tool-guard"; +import type { AdapterEvent } from "../../src/types"; import { isCodeModeMcpDirectName, normalizeDeclaredToolName } from "../../src/types/tools"; +import { dataPayload, frame } from "../helpers/custom-tool-repair-fixtures"; const CODE_MODE = new Set(["exec"]); +const MCP_NAME = "mcp__codex_app__get_usage_limits"; +const CALL = { type: "function_call", id: "fc_mcp", call_id: "call_mcp", name: MCP_NAME, arguments: "{}" }; +const CUSTOM_EXEC = { type: "namespace", name: "functions", tools: [{ type: "custom", name: "exec", format: { type: "text" } }] }; +const FUNCTION_EXEC = { type: "namespace", name: "functions", tools: [{ type: "function", name: "exec", parameters: { type: "object" } }] }; +const FOREIGN_EXEC = { type: "namespace", name: "mcp__remote", tools: [{ type: "custom", name: "exec", format: { type: "text" } }] }; + +function mapsFor(tools: unknown[]) { + const parsed = parseRequest({ model: "test-model", input: "run it", tools }); + return buildToolBridgeMaps(parsed); +} + +function eventsFor(name: string): AdapterEvent[] { + return [ + { type: "tool_call_start", id: "call_mcp", name }, + { type: "tool_call_delta", id: "call_mcp", arguments: "{}" }, + { type: "tool_call_end", id: "call_mcp" }, + { type: "done" }, + ]; +} // Codex Desktop code mode declares only the freeform `exec` shell; every nested host tool // (`tools.mcp__codex_app__get_usage_limits`, ...) is reachable through it but undeclared. @@ -21,21 +50,53 @@ describe("code-mode direct mcp tool-call recovery", () => { } }); - test("maps a direct mcp call through a declared exec only", () => { - expect(normalizeDeclaredToolName("mcp__codex_app__get_usage_limits", CODE_MODE)).toBe("exec"); + test("maps a direct mcp call only through an explicitly custom exec", () => { + expect(normalizeDeclaredToolName(MCP_NAME, CODE_MODE, undefined, CODE_MODE)).toBe("exec"); + expect(normalizeDeclaredToolName(`default.${MCP_NAME}`, CODE_MODE, undefined, CODE_MODE)).toBe("exec"); + expect(normalizeDeclaredToolName(MCP_NAME, CODE_MODE)).toBe(MCP_NAME); + expect(normalizeDeclaredToolName(`default.${MCP_NAME}`, CODE_MODE)).toBe(`default.${MCP_NAME}`); expect(normalizeDeclaredToolName("mcp__codex_app__get_usage_limits", new Set())).toBe("mcp__codex_app__get_usage_limits"); // A catalog that declares the name itself keeps the call's own identity. expect(normalizeDeclaredToolName( "mcp__codex_app__get_usage_limits", - new Set(["exec", "mcp__codex_app__get_usage_limits"]), + new Set(["exec", MCP_NAME]), undefined, CODE_MODE, )).toBe("mcp__codex_app__get_usage_limits"); // The flat-bridge shape (legacy shell names declared next to exec) is not code mode. expect(normalizeDeclaredToolName( "mcp__codex_app__get_usage_limits", - new Set(["exec", "exec_command"]), + new Set(["exec", "exec_command"]), undefined, CODE_MODE, )).toBe("mcp__codex_app__get_usage_limits"); // Malformed mcp-ish names stay undeclared. - expect(normalizeDeclaredToolName("mcp__server", CODE_MODE)).toBe("mcp__server"); + expect(normalizeDeclaredToolName("mcp__server", CODE_MODE, undefined, CODE_MODE)).toBe("mcp__server"); + expect(normalizeDeclaredToolName(`default.${MCP_NAME}`, new Set(["exec", `default.${MCP_NAME}`]), undefined, CODE_MODE)) + .toBe(`default.${MCP_NAME}`); + }); + + test("catalog provenance excludes JSON functions and foreign namespace aliases", () => { + const custom = mapsFor([CUSTOM_EXEC]); + const ordinary = mapsFor([FUNCTION_EXEC]); + const foreign = mapsFor([FOREIGN_EXEC]); + expect(custom.bareCustomToolNames).toEqual(CODE_MODE); + expect(ordinary.declaredToolNames.has("exec")).toBe(true); + expect(ordinary.bareCustomToolNames.has("exec")).toBe(false); + expect(foreign.declaredToolNames.has("exec")).toBe(false); + expect(foreign.bareCustomToolNames.has("exec")).toBe(false); + expect(foreign.declaredToolNames.has("mcp__remote__exec")).toBe(true); + for (const maps of [ordinary, foreign]) { + expect(normalizeDeclaredToolName(MCP_NAME, maps.declaredToolNames, undefined, maps.bareCustomToolNames)).toBe(MCP_NAME); + } + }); + + test("an explicitly declared MCP function keeps its namespaced identity", () => { + const maps = mapsFor([CUSTOM_EXEC, { + type: "namespace", name: "mcp__codex_app", + tools: [{ type: "function", name: "get_usage_limits", parameters: { type: "object" } }], + }]); + expect(buildResponseJSON(eventsFor(MCP_NAME), "fixture", { + ...maps, enforceDeclaredToolNames: true, + }).output).toMatchObject([{ + type: "function_call", name: "get_usage_limits", namespace: "mcp__codex_app", arguments: "{}", + }]); }); test("compiles the call to the matching nested host tool", () => { @@ -50,37 +111,49 @@ describe("code-mode direct mcp tool-call recovery", () => { expect(compileCodeModeHelperInput("not json", "mcp__x__y")).toBe( 'const result = await tools.mcp__x__y("not json");\ntext(result);', ); + expect(compileCodeModeHelperInput("{}", `default.${MCP_NAME}`)).toBe( + `const result = await tools.${MCP_NAME}({});\ntext(result);`, + ); }); - test("keeps the undeclared-tool guard admit/block boundary unchanged elsewhere", () => { - const source = { - output: [{ - type: "function_call", - id: "fc_mcp", - call_id: "call_mcp", - name: "mcp__codex_app__get_usage_limits", - arguments: "{}", - }], - }; - expect(undeclaredToolCallNameInResponse(source, CODE_MODE)).toBeUndefined(); - expect(undeclaredToolCallNameInResponse(source, new Set())).toBe("mcp__codex_app__get_usage_limits"); + test("generated JavaScript uses host tool lookup and keeps arguments as data", async () => { + const args = { query: '"; throw new Error("injected") //', limit: 3 }; + const input = compileCodeModeHelperInput(JSON.stringify(args), MCP_NAME); + const run = new Function("tools", "text", `return (async () => { ${input} })();`); + const calls: unknown[] = []; + const output: unknown[] = []; + await run({ [MCP_NAME]: async (value: unknown) => { calls.push(value); return "ok"; } }, (value: unknown) => output.push(value)); + expect(calls).toEqual([args]); + expect(output).toEqual(["ok"]); + await expect(run({}, () => {})).rejects.toThrow(); + }); + + test("guard admits direct calls only with current bare custom provenance", () => { + const source = { output: [CALL] }; + expect(undeclaredToolCallNameInResponse(source, CODE_MODE, undefined, undefined, undefined, CODE_MODE)).toBeUndefined(); + expect(undeclaredToolCallNameInResponse( + { output: [{ ...CALL, name: `default.${MCP_NAME}` }] }, CODE_MODE, + undefined, undefined, undefined, CODE_MODE, + )).toBeUndefined(); + expect(undeclaredToolCallNameInResponse(source, CODE_MODE)).toBe(MCP_NAME); + expect(undeclaredToolCallNameInResponse(source, new Set())).toBe(MCP_NAME); + const current = { tools: [FUNCTION_EXEC], input: [{ type: "additional_tools", tools: [CUSTOM_EXEC] }] }; + expect(collectDeclaredBareCustomWireToolNames(current)).toEqual(CODE_MODE); + expect(collectDeclaredBareCustomWireToolNames(currentTurnWireToolCatalogBody(current, 1))).toEqual(new Set()); + expect(collectDeclaredBareCustomWireToolNames({ tools: [FOREIGN_EXEC] })).toEqual(new Set()); + expect(undeclaredToolCallNameInResponse( + { output: [{ ...CALL, name: "get_usage_limits", namespace: "mcp__codex_app" }] }, + CODE_MODE, undefined, undefined, undefined, CODE_MODE, + )).toBe("get_usage_limits"); // A name that only looks mcp-ish is still blocked under a code-mode catalog. const malformed = { output: [{ type: "function_call", id: "fc_x", call_id: "call_x", name: "mcp__solo", arguments: "{}" }], }; - expect(undeclaredToolCallNameInResponse(malformed, CODE_MODE)).toBe("mcp__solo"); + expect(undeclaredToolCallNameInResponse(malformed, CODE_MODE, undefined, undefined, undefined, CODE_MODE)).toBe("mcp__solo"); }); test("restores a recorded direct mcp call as the declared exec", () => { - const source = { - output: [{ - type: "function_call", - id: "fc_mcp", - call_id: "call_mcp", - name: "mcp__codex_app__get_usage_limits", - arguments: "{}", - }], - }; + const source = { output: [CALL] }; const restored = JSON.parse(restoreRoutedCustomCallsInJson( JSON.stringify(source), CODE_MODE, @@ -95,5 +168,57 @@ describe("code-mode direct mcp tool-call recovery", () => { }]); expect(undeclaredToolCallNameInResponse(restored, CODE_MODE)).toBeUndefined(); }); -}); + test("native SSE restoration compiles only a converted custom exec call", () => { + for (const name of [MCP_NAME, `default.${MCP_NAME}`]) { + const customNames = rewriteRoutedCustomToolsForUpstream({ tools: [CUSTOM_EXEC] }, false).names; + const rewrite = createRoutedCustomToolRestoreBlockRewrite(customNames, undefined, new Set(), CODE_MODE); + const added = rewrite(frame("response.output_item.added", { + output_index: 0, + item: { ...CALL, id: "fc_native", name, arguments: "", status: "in_progress" }, + })); + expect(dataPayload(added[0]!).item).toMatchObject({ type: "custom_tool_call", name: "exec" }); + const done = rewrite(frame("response.function_call_arguments.done", { + output_index: 0, item_id: "fc_native", arguments: "{}", + })); + expect(dataPayload(done[0]!).input).toBe(`const result = await tools.${MCP_NAME}({});\ntext(result);`); + rewrite.dispose?.(); + } + }); + + test("JSON, SSE and historical restoration agree on custom, ordinary and foreign exec", async () => { + for (const [tools, accepted] of [ + [[CUSTOM_EXEC], true], + [[FUNCTION_EXEC], false], + [[FOREIGN_EXEC], false], + ] as const) { + const maps = mapsFor([...tools]); + const options = { ...maps, enforceDeclaredToolNames: true }; + for (const name of [MCP_NAME, `default.${MCP_NAME}`]) { + const expectedInput = `const result = await tools.${MCP_NAME}({});\ntext(result);`; + const json = buildResponseJSON(eventsFor(name), "fixture", options); + async function* streamEvents(): AsyncGenerator { yield* eventsFor(name); } + const stream = bridgeToResponsesSSE( + streamEvents(), "fixture", maps.toolNsMap, maps.freeformToolNames, + maps.toolSearchToolNames, undefined, 50_000, options, + ); + const text = await new Response(stream).text(); + const payloads = text.split(/\r?\n\r?\n/).filter(block => block.includes("data: {")).map(dataPayload); + const historical = JSON.parse(restoreRoutedCustomCallsInJson( + JSON.stringify({ output: [{ ...CALL, name }] }), + rewriteRoutedCustomToolsForUpstream({ tools }, false).names, + new Set(), maps.declaredToolNames, + )) as { output: Array> }; + if (accepted) { + expect(json.output).toMatchObject([{ type: "custom_tool_call", name: "exec", input: expectedInput }]); + expect(payloads.find(p => p.type === "response.custom_tool_call_input.done")?.input).toBe(expectedInput); + expect(historical.output[0]).toMatchObject({ type: "custom_tool_call", name: "exec", input: expectedInput }); + } else { + expect(json.status).toBe("failed"); + expect(payloads.some(p => p.type === "response.failed")).toBe(true); + expect(historical.output[0]).toMatchObject({ type: "function_call", name }); + } + } + } + }); +}); From 23444c1bec2afe51feccd976ae0c74d17d4eca7f Mon Sep 17 00:00:00 2001 From: mdwsk88 <924038395@qq.com> Date: Sun, 27 Sep 2026 10:51:27 +0800 Subject: [PATCH 3/4] fix(responses): pass custom provenance to client encoder --- .../inference/client-encoder-delivery.ts | 1 + src/server/responses/adapter-delivery.ts | 2 +- .../inference-client-encoder-delivery.test.ts | 38 +++++++++++++++++++ 3 files changed, 40 insertions(+), 1 deletion(-) diff --git a/src/server/inference/client-encoder-delivery.ts b/src/server/inference/client-encoder-delivery.ts index 1dc2a649e71..5cd6d1c22c2 100644 --- a/src/server/inference/client-encoder-delivery.ts +++ b/src/server/inference/client-encoder-delivery.ts @@ -91,6 +91,7 @@ export interface ClientEncodedDelivery { declaredToolNames?: ReadonlySet; toolParameterSchemas?: ReadonlyMap>; freeformToolNames?: Set; + bareCustomToolNames?: ReadonlySet; toolSearchToolNames?: Set; }; stallTimeoutSec?: number; diff --git a/src/server/responses/adapter-delivery.ts b/src/server/responses/adapter-delivery.ts index f327e0137a3..321783be985 100644 --- a/src/server/responses/adapter-delivery.ts +++ b/src/server/responses/adapter-delivery.ts @@ -153,7 +153,7 @@ export async function deliverAdapterResponse( fold: { replayCacheScope: parsed._reasoningReplayScope, hideThinkingSummary: parsed.options.hideThinkingSummary, - toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames, + toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames, }, stallTimeoutSec: config.stallTimeoutSec, localUpstream, diff --git a/tests/server/inference-client-encoder-delivery.test.ts b/tests/server/inference-client-encoder-delivery.test.ts index f2aa53baaca..510827260a2 100644 --- a/tests/server/inference-client-encoder-delivery.test.ts +++ b/tests/server/inference-client-encoder-delivery.test.ts @@ -184,4 +184,42 @@ describe("deliverClientEncodedResponse", () => { expect(body.choices[0]!.message.content).toBe("hi"); expect(run.completed).toHaveLength(1); }); + + test("the client-encoder fold preserves code-mode direct MCP recovery", async () => { + const logCtx: RequestLogContext = { model: "m", provider: "p" }; + const toolEvents: AdapterEvent[] = [ + { type: "tool_call_start", id: "call_mcp", name: "mcp__codex_app__get_usage_limits" }, + { type: "tool_call_delta", id: "call_mcp", arguments: "{}" }, + { type: "tool_call_end", id: "call_mcp" }, + { type: "done" }, + ]; + const completed: Record[] = []; + const input = { + encoder: { protocol: "chat" as const, stream: false, model: "client-model" }, + events: replay(toolEvents), + logCtx, + translatorBudget: createTestTranslatorBudget(), + responseModelId: "internal/model", + adapterName: "anthropic", + fold: { + declaredToolNames: new Set(["exec"]), + freeformToolNames: new Set(["exec"]), + bareCustomToolNames: new Set(["exec"]), + }, + stopUpstream: () => {}, + onStreamDone: () => {}, + onCompletedResponse: (response: Record) => { completed.push(response); }, + bindUsage: () => {}, + }; + const response = await deliverClientEncodedResponse(input); + expect(response.status).toBe(200); + expect(completed).toHaveLength(1); + expect(completed[0]).toMatchObject({ + output: [{ + type: "custom_tool_call", + name: "exec", + input: 'const result = await tools.mcp__codex_app__get_usage_limits({});\ntext(result);', + }], + }); + }); }); From 06fa8855ba72b61787847fabb5945fa90aa938e5 Mon Sep 17 00:00:00 2001 From: mdwsk88 <924038395@qq.com> Date: Sun, 27 Sep 2026 13:30:59 +0800 Subject: [PATCH 4/4] test(responses): drive the client-encoder regression through adapter delivery The previous regression handed a prebuilt fold to deliverClientEncodedResponse, so it passed with or without the adapter-delivery wiring it was meant to pin. Drive deliverAdapterResponse with a parsed code-mode request instead: the test now fails when bareCustomToolNames is dropped from the encoder fold. Co-Authored-By: Claude Opus 5.5 --- .../inference-client-encoder-delivery.test.ts | 74 ++++++++++++++----- 1 file changed, 54 insertions(+), 20 deletions(-) diff --git a/tests/server/inference-client-encoder-delivery.test.ts b/tests/server/inference-client-encoder-delivery.test.ts index 510827260a2..e11a4d9dff9 100644 --- a/tests/server/inference-client-encoder-delivery.test.ts +++ b/tests/server/inference-client-encoder-delivery.test.ts @@ -16,6 +16,9 @@ import { beginInferenceAttempt } from "../../src/server/inference/attempt"; import { responseWithDeferredRequestLog } from "../../src/server/relay"; import type { RequestLogContext, RequestLogEntry } from "../../src/server/request-log"; import { markProtocolEntry, protocolTraceForRequest } from "../../src/protocols/trace"; +import { parseRequest } from "../../src/responses/parser"; +import { buildToolBridgeMaps } from "../../src/server/responses"; +import { deliverAdapterResponse } from "../../src/server/responses/adapter-delivery"; import type { AdapterEvent, OcxConfig, OcxUsage } from "../../src/types"; import { createTestTranslatorBudget } from "../helpers/translator-budget"; @@ -185,8 +188,24 @@ describe("deliverClientEncodedResponse", () => { expect(run.completed).toHaveLength(1); }); - test("the client-encoder fold preserves code-mode direct MCP recovery", async () => { - const logCtx: RequestLogContext = { model: "m", provider: "p" }; +}); + +// The routed-adapter branch assembles the encoder's fold itself. Direct MCP recovery reaches the +// completed response only when that fold carries the request's custom exec provenance, so this +// drives the real delivery entry point rather than a hand-built fold. +describe("deliverAdapterResponse through a client encoder", () => { + type Args = Parameters; + + test("the fold keeps code-mode direct MCP recovery", async () => { + const parsed = parseRequest({ + model: "internal/model", + input: "check usage", + stream: true, + store: false, + tools: [{ type: "namespace", name: "functions", tools: [{ type: "custom", name: "exec", format: { type: "text" } }] }], + }); + const toolBridgeMaps = buildToolBridgeMaps(parsed); + expect(toolBridgeMaps.bareCustomToolNames).toEqual(new Set(["exec"])); const toolEvents: AdapterEvent[] = [ { type: "tool_call_start", id: "call_mcp", name: "mcp__codex_app__get_usage_limits" }, { type: "tool_call_delta", id: "call_mcp", arguments: "{}" }, @@ -194,24 +213,39 @@ describe("deliverClientEncodedResponse", () => { { type: "done" }, ]; const completed: Record[] = []; - const input = { - encoder: { protocol: "chat" as const, stream: false, model: "client-model" }, - events: replay(toolEvents), - logCtx, - translatorBudget: createTestTranslatorBudget(), - responseModelId: "internal/model", - adapterName: "anthropic", - fold: { - declaredToolNames: new Set(["exec"]), - freeformToolNames: new Set(["exec"]), - bareCustomToolNames: new Set(["exec"]), - }, - stopUpstream: () => {}, - onStreamDone: () => {}, - onCompletedResponse: (response: Record) => { completed.push(response); }, - bindUsage: () => {}, - }; - const response = await deliverClientEncodedResponse(input); + const response = await deliverAdapterResponse( + { + logCtx: { model: "m", provider: "p" }, + options: { clientEncoder: { protocol: "chat", stream: false, model: "client-model" } }, + config: {}, + } as unknown as Args[0], + { + parsed, + translatorBudget: createTestTranslatorBudget(), + toolBridgeMaps, + rememberKiroDeliveredFinalAnswer: () => {}, + responseStateOptions: () => ({}), + } as unknown as Args[1], + { + activeAdapter: { name: "anthropic", parseStream: () => replay(toolEvents) }, + bindKeyUsageFromBridge: () => {}, + } as unknown as Args[2], + { routedCompaction: undefined } as unknown as Args[3], + { + cancelResponseCompletion: () => {}, + commitReasoningReplayServingRoute: () => {}, + continuationStateForResponse: () => undefined, + notifyResponseComplete: (folded: Record) => { completed.push(folded); }, + } as unknown as Args[4], + { emptyCompletionGuardEnabled: false }, + { + upstreamResponse: new Response(""), + upstream: new AbortController(), + cleanupUpstreamAbort: () => {}, + localUpstream: false, + } as unknown as Args[6], + { terminalGuardEnabled: false } as unknown as Args[7], + ); expect(response.status).toBe(200); expect(completed).toHaveLength(1); expect(completed[0]).toMatchObject({