From 90dccd8d15f66be4159446b197c84a7964823bd7 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Fri, 9 Oct 2026 02:25:59 +0000 Subject: [PATCH 1/4] fix native structured output projection --- contracts/agents-api/execution-tools.md | 2 +- contracts/agents-api/index.md | 1 + contracts/agents-api/zh/execution-tools.md | 4 +- contracts/agents-api/zh/index.md | 3 +- .../src/structured_output.ts | 84 +++++++++--- .../tests/executor.test.mjs | 49 ++++++- .../tests/structured_output.test.mjs | 123 +++++++++++++++--- 7 files changed, 226 insertions(+), 40 deletions(-) diff --git a/contracts/agents-api/execution-tools.md b/contracts/agents-api/execution-tools.md index eefe52fc7..6ddfd2391 100644 --- a/contracts/agents-api/execution-tools.md +++ b/contracts/agents-api/execution-tools.md @@ -41,7 +41,7 @@ The Worker fails previously claimed work without replay after execution loss; qu Core carries the schema in `ExecutionControls.OutputFormat` and requires `structured_output` in the Harness's declaration and the Runtime's heartbeat, only for requests that use the option. Frozen schemas reach preparation before input and apply to initial and resumed execution; Start cannot replace them. -The Claude adapter passes `outputFormat` to the pinned SDK and allows its native `StructuredOutput` terminal tool, which is internal and never an extra caller function. A matching live root tool result and an attributed successful SDK result confirm the output. The adapter publishes the native `result.result` string unchanged as a completed `final_answer` message with the native tool-use ID; parent assistant prose keeps its own ID. Unvalidated retries and cancelled candidates never become the answer, and the adapter never serializes `structured_output` back to JSON. The stream follows the official message sequence with the whole text in one `output_text.delta`. The bridge advertises the operation only when it reports `structured_output`, and a workspace Runtime also needs `workspace_structured_output`. +The Claude adapter passes `outputFormat` to the pinned SDK and allows its native `StructuredOutput` terminal tool, which is internal and never an extra caller function. The adapter accumulates the original `input_json_delta` text by live root Session, native message ID, content-block index and tool-use ID. Only a completed matching assistant tool call, its successful receipt and a unique candidate consistent with the attributed successful SDK `structured_output` confirm that text. The adapter publishes it unchanged as a completed `final_answer` message with the native tool-use ID; parent assistant prose keeps its own ID. Unvalidated retries, missing or ambiguous identities, retracted work and cancelled candidates never become the answer. The generic SDK `result` text is not the structured payload, and the adapter never serializes `structured_output` back to JSON. Raw numeric digits are preserved; exact-number schema validation remains a [known gap](./index.md#known-gaps). The stream follows the official message sequence with the whole text in one `output_text.delta`. The bridge advertises the operation only when it reports `structured_output`, and a workspace Runtime also needs `workspace_structured_output`. ## Deferred function discovery diff --git a/contracts/agents-api/index.md b/contracts/agents-api/index.md index bdbbf26c9..f862272bd 100644 --- a/contracts/agents-api/index.md +++ b/contracts/agents-api/index.md @@ -110,6 +110,7 @@ Each item is Core's deliberate or native behavior where the official service beh - Explicit reasoning effort or summary, service tiers other than `auto`, enabled `web_search` and enabled programmatic tool calling are saved but rejected at Session admission. - Harness support differs as each [declaration](./harness-onboarding.md#declare-support) states. Codex has no structured output or `tool_search`. Claude Code takes no whitespace-only text, `medium` verbosity only and function-result images only inline and in successful results; it rejects Subagents with MCP, including MCP servers that Environment Plugins install; structured output with Subagents, MCP, installed capabilities or `tool_search`; and `tool_search` with MCP or installed capabilities. MiniMax Code has no public functions, no service-origin MCP, no image input, no whitespace-only text, no required MCP and `medium` verbosity only, and takes `allowed_tools` null only. Each Harness reserves an MCP `server_label`: `codex_apps` for Codex, `functions` for Claude Code and `oac_workspace` for MiniMax Code. Claude Code also requires labels to match `^[a-zA-Z0-9_-]+$` and `allowed_tools` names to match `^[a-zA-Z0-9_.-]+$`. +- Claude structured output preserves the [acknowledged raw tool input](./execution-tools.md#structured-output), while native validation and parsed-value correlation use binary64. Distinct raw numbers can compare equal after parsing, so this does not establish exact-number schema conformance, even when every schema literal passes admission. Exact numeric output validation remains unqualified. - Model-derived reasoning defaults are not resolved. - MCP tools support the `http` transport only; `stdio` is rejected, and so is an inline `authorization` on a Session MCP transport ([HTTP MCP](./execution-tools.md#http-mcp)). diff --git a/contracts/agents-api/zh/execution-tools.md b/contracts/agents-api/zh/execution-tools.md index 9c4692a52..2f31fe5cc 100644 --- a/contracts/agents-api/zh/execution-tools.md +++ b/contracts/agents-api/zh/execution-tools.md @@ -1,7 +1,7 @@ --- title: "执行工具" source: contracts/agents-api/execution-tools.md -source_hash: 21ba998ee2697491899ab001367ab7dd2dea7ef86f4bfc015b95cbbf5beba558 +source_hash: fbd6a39ac815c0e3af66f72d3690c8ab7b705f45a8f5eedbdf6c00a33eebe355 --- Agent 在 `tools` 中声明应用函数、控制项和 MCP 服务器,并可在 `text.format` 中声明输出 schema。本契约说明 Core 如何验证声明、哪些内容跨越 Runtime 边界,以及调用方如何恢复待执行操作。每个 Harness 的[声明](harness-onboarding.md#declare-support)说明它支持其中哪些内容。原生工作区工具和 Environment Plugin MCP 属于 [Environment](environments.md#skills-plugins-and-environment-mcp)。 @@ -43,7 +43,7 @@ SSE 仅提供实时事件。重启或流丢失后,读取 Session 的 `required Core 在 `ExecutionControls.OutputFormat` 中携带 schema,仅对使用该选项的请求要求 Harness 的声明和 Runtime 的心跳都支持 `structured_output`。冻结的 schema 在输入之前送达准备阶段,适用于初次和恢复执行;Start 不能替换它。 -Claude 适配器将 `outputFormat` 传给固定版本 SDK,并允许原生 `StructuredOutput` 终态工具;该工具属于内部,不是额外的调用方函数。匹配的实时根工具结果和已归属的成功 SDK 结果确认输出。适配器将原生 `result.result` 字符串原样发布为已完成的 `final_answer` 消息,使用原生 tool-use ID;父 assistant 文本保留自己的 ID。未验证重试和已取消候选不会成为答案,适配器不会把 `structured_output` 重新序列化为 JSON。流遵循官方消息顺序,将整段文本放入一个 `output_text.delta`。桥接层仅在报告 `structured_output` 时声明该操作,工作区 Runtime 还需要 `workspace_structured_output`。 +Claude 适配器将 `outputFormat` 传给固定版本 SDK,并允许原生 `StructuredOutput` 终态工具;该工具属于内部,不是额外的调用方函数。适配器按实时根 Session、原生消息 ID、内容块索引和 tool-use ID 累积原始 `input_json_delta` 文本。只有匹配的完整 assistant 工具调用、成功回执,以及与已归属的成功 SDK `structured_output` 一致的唯一候选,才能确认该文本。适配器将其原样发布为已完成的 `final_answer` 消息,使用原生 tool-use ID;父 assistant 文本保留自己的 ID。未验证重试、缺失或歧义身份、已撤回工作和已取消候选不会成为答案。通用 SDK `result` 文本不是结构化载荷,适配器不会把 `structured_output` 重新序列化为 JSON。原始数字文本会被保留;精确数值 schema 验证仍是[已知缺口](./index.md#known-gaps)。流遵循官方消息顺序,将整段文本放入一个 `output_text.delta`。桥接层仅在报告 `structured_output` 时声明该操作,工作区 Runtime 还需要 `workspace_structured_output`。 ## 延迟函数发现 {#deferred-function-discovery} diff --git a/contracts/agents-api/zh/index.md b/contracts/agents-api/zh/index.md index 80da43e72..b35444d53 100644 --- a/contracts/agents-api/zh/index.md +++ b/contracts/agents-api/zh/index.md @@ -1,7 +1,7 @@ --- title: "Agents API 覆盖台账" source: contracts/agents-api/index.md -source_hash: 480884468c15159c84b123978c1d8d5fdac89bd3833556100be1ef4d31fb8829 +source_hash: 34f04869ed4a0cf01754d7375688c7176083eff5ec26f68321adbace76158be9 --- Core 旨在以下方固定版本为准支持完整的 OpenAI Agents API([public API rule](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/AGENTS.md#public-api))。本台账记录 Core 对各项资源实现了哪些内容、哪些契约保存其详细信息,并列出相对于 OpenAI 服务的所有已知差异和所有未解决缺口。[API namespaces and credentials](../../../docs/zh/api/index.md) 说明谁调用哪些 API;[Agents API guide](../../../docs/zh/api/public-agent-api.md) 介绍使用方法。 @@ -112,6 +112,7 @@ Core 自身字段位于 `x_agents_core` 中([Core extensions](../../../docs/zh - 显式指定推理强度或摘要、使用 `auto` 之外的服务层级、启用 `web_search` 或启用程序化工具调用,这些设置都会被保存,但在 Session 准入时会被拒绝。 - 各 Harness 的支持差异以其[声明](harness-onboarding.md#declare-support)为准。Codex 不支持结构化输出或 `tool_search`。Claude Code 不接受仅含空白的文本,只支持 `medium` 详细程度,函数结果图像只能内联且只能出现在成功结果中;它拒绝子智能体与 MCP(包括 Environment Plugins 安装的 MCP server)同时使用,拒绝结构化输出与子智能体、MCP、已安装能力或 `tool_search` 同时使用,也拒绝 `tool_search` 与 MCP 或已安装能力同时使用。MiniMax Code 不提供公共 functions,没有服务源 MCP,不支持图像输入、仅含空白的文本和必需 MCP,只支持 `medium` 详细程度,且只接受值为 null 的 `allowed_tools`。每个 Harness 都保留一个 MCP `server_label`:Codex 保留 `codex_apps`,Claude Code 保留 `functions`,MiniMax Code 保留 `oac_workspace`。Claude Code 还要求标签匹配 `^[a-zA-Z0-9_-]+$`,`allowed_tools` 中的名称匹配 `^[a-zA-Z0-9_.-]+$`。 +- Claude 结构化输出保留[已确认的原始工具输入](./execution-tools.md#structured-output),但原生验证和解析值关联使用 binary64。不同的原始数字在解析后可能相等,因此即使 schema 的所有数值字面量都通过准入,也不能据此确认精确数值的 schema 符合性。精确数值输出验证仍未经资格验证。 - 由模型推导出的推理默认值不会被解析确定。 - MCP 工具仅支持 `http` 传输,`stdio` 会被拒绝,Session MCP 传输中的内联 `authorization` 也会被拒绝([HTTP MCP](execution-tools.md#http-mcp))。 diff --git a/packages/claude-sdk-adapter/src/structured_output.ts b/packages/claude-sdk-adapter/src/structured_output.ts index e39c28c1e..bf31429f6 100644 --- a/packages/claude-sdk-adapter/src/structured_output.ts +++ b/packages/claude-sdk-adapter/src/structured_output.ts @@ -1,39 +1,91 @@ import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk"; +import { isDeepStrictEqual } from "node:util"; import type { MessageEvent } from "./messages.js"; +type Candidate = { id: string; text: string; snapshot?: string; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "published" }; + // StructuredOutput is the native terminal tool. Its acknowledged tool-use identity // owns the final JSON message; the parent assistant may already own ordinary prose. export class StructuredOutput { - private readonly calls = new Set(); - private accepted?: string; + private readonly calls = new Map(); + private readonly messages = new Set(); + private readonly events = new Set(); + private active?: { id: string; blocks: Map }; + private session = ""; consume(message: SDKMessage, session: string): void { if (!session || message.session_id !== session || - !("parent_tool_use_id" in message) || message.parent_tool_use_id !== null || ("isReplay" in message && message.isReplay) || ("isSynthetic" in message && message.isSynthetic)) return; - if (message.type === "assistant") { + const retracted = message.type === "assistant" ? message.supersedes : + message.type === "system" && message.subtype === "model_refusal_fallback" ? message.retracted_message_uuids : undefined; + if (retracted?.some(id => [...this.calls.values()].some(call => call.snapshot === id || call.receipt === id))) { + throw new Error("retracted structured output"); + } + if (!("parent_tool_use_id" in message) || message.parent_tool_use_id !== null) return; + this.session = session; + if (message.type === "stream_event") { + if (!message.uuid || this.events.has(message.uuid)) throw new Error("invalid structured stream identity"); + this.events.add(message.uuid); + const event = message.event; + if (event.type === "message_start") { + if (!event.message.id || this.messages.has(event.message.id) || this.active?.blocks.size) throw new Error("invalid structured message identity"); + this.messages.add(event.message.id); + this.active = { id: event.message.id, blocks: new Map() }; + } else if (event.type === "content_block_start" && event.content_block.type === "tool_use" && event.content_block.name === "StructuredOutput") { + const id = event.content_block.id; + if (!this.active || !id || this.calls.has(id) || this.active.blocks.has(event.index) || + [...this.calls.values()].some(call => call.status === "accepted")) throw new Error("invalid structured output identity"); + const call: Candidate = { id, text: "", stopped: false }; + this.calls.set(id, call); + this.active.blocks.set(event.index, call); + } else if (event.type === "content_block_delta" && event.delta.type === "input_json_delta") { + const call = this.active?.blocks.get(event.index); + if (call) { + if (call.snapshot || call.stopped) throw new Error("late structured output input"); + call.text += event.delta.partial_json; + } + } else if (event.type === "content_block_stop") { + const call = this.active?.blocks.get(event.index); + if (call) { + if (!call.snapshot || call.stopped) throw new Error("unconfirmed structured output block"); + call.stopped = true; + } + } else if (event.type === "message_stop") { + if (this.active && [...this.active.blocks.values()].some(call => !call.stopped)) throw new Error("incomplete structured output message"); + this.active = undefined; + } + } else if (message.type === "assistant") { for (const block of message.message.content) { if (block.type !== "tool_use" || block.name !== "StructuredOutput") continue; - if (!block.id || this.calls.has(block.id)) throw new Error("invalid structured output identity"); - this.calls.add(block.id); + const call = this.calls.get(block.id); + if (!call || !this.active || message.message.id !== this.active.id || + ![...this.active.blocks.values()].includes(call) || call.snapshot || !message.uuid || message.error || message.aborted || + !isDeepStrictEqual(JSON.parse(call.text), block.input)) throw new Error("unmatched structured output snapshot"); + call.snapshot = message.uuid; } } else if (message.type === "user" && Array.isArray(message.message.content)) { for (const block of message.message.content) { - if (block.type !== "tool_result" || !this.calls.delete(block.tool_use_id)) continue; - if (!block.is_error) this.accepted = block.tool_use_id; + if (block.type !== "tool_result") continue; + const call = this.calls.get(block.tool_use_id); + if (!call) continue; + if (!call.snapshot || call.receipt || !message.uuid) throw new Error("invalid structured output receipt"); + call.receipt = message.uuid; + call.status = block.is_error ? "failed" : "accepted"; } } } complete(message: SDKMessage): MessageEvent { + const calls = [...this.calls.values()]; + const accepted = calls.filter(call => call.status === "accepted"); if (message.type !== "result" || message.subtype !== "success" || message.is_error || - message.structured_output === undefined || !this.accepted || this.calls.size || - typeof message.result !== "string") throw new Error("unconfirmed structured output"); - // Preserve the SDK's final string. Reserializing structured_output can round - // JSON numbers and would replace the native validated result. - JSON.parse(message.result); - const id = this.accepted; - this.accepted = undefined; - return { type: "output_message", message: { id, status: "completed", phase: "final_answer", text: message.result } }; + message.session_id !== this.session || message.structured_output === undefined || this.active?.blocks.size || + calls.some(call => !call.stopped || !call.receipt) || accepted.length !== 1 || + !isDeepStrictEqual(JSON.parse(accepted[0]!.text), message.structured_output)) throw new Error("unconfirmed structured output"); + // Native validation compares binary64 values. Publish only the attributed raw + // tool input: reserializing the validated object would lose original digits. + const { id, text } = accepted[0]!; + accepted[0]!.status = "published"; + return { type: "output_message", message: { id, status: "completed", phase: "final_answer", text } }; } } diff --git a/packages/claude-sdk-adapter/tests/executor.test.mjs b/packages/claude-sdk-adapter/tests/executor.test.mjs index a9fe08eda..97778816f 100644 --- a/packages/claude-sdk-adapter/tests/executor.test.mjs +++ b/packages/claude-sdk-adapter/tests/executor.test.mjs @@ -30,7 +30,7 @@ globalThis.startupFixture=async({options})=>{ number++; const text=user.message.content[0].text; process.send({type:'input',text}); - yield {type:'system',subtype:'init',session_id:'native',tools:client ? ['mcp__functions__lookup'] : [],mcp_servers:client ? [{name:'functions',status:'connected'}] : []}; + yield {type:'system',subtype:'init',session_id:'native',tools:client ? ['mcp__functions__lookup'] : options.outputFormat ? ['StructuredOutput'] : [],mcp_servers:client ? [{name:'functions',status:'connected'}] : []}; if(client){ if(text==='pending-function'){ void client.callTool({name:'lookup',arguments:{text},_meta:{'claudecode/toolUseId':'pending-call'}}).catch(()=>{}); @@ -47,14 +47,32 @@ globalThis.startupFixture=async({options})=>{ {type:'content_block_stop',index:0},{type:'message_stop'} ])yield {type:'stream_event',parent_tool_use_id:null,session_id:'native',event}; } + if(options.outputFormat){ + const raw='{"number":9007199254740993}'; + for(const event of [ + {type:'message_start',message:{id:'structured-message'}}, + {type:'content_block_start',index:0,content_block:{type:'text',text:'ordinary prose'}}, + {type:'content_block_stop',index:0}, + {type:'content_block_start',index:1,content_block:{type:'tool_use',id:'structured-call',name:'StructuredOutput',input:{}}}, + {type:'content_block_delta',index:1,delta:{type:'input_json_delta',partial_json:raw}} + ])yield {type:'stream_event',uuid:crypto.randomUUID(),parent_tool_use_id:null,session_id:'native',event}; + yield {type:'assistant',uuid:'structured-snapshot',session_id:'native',parent_tool_use_id:null,message:{id:'structured-message',content:[{type:'tool_use',id:'structured-call',name:'StructuredOutput',input:JSON.parse(raw)}]}}; + for(const event of [{type:'content_block_stop',index:1},{type:'message_stop'}])yield {type:'stream_event',uuid:crypto.randomUUID(),parent_tool_use_id:null,session_id:'native',event}; + yield {type:'user',uuid:'structured-receipt',session_id:'native',parent_tool_use_id:null,message:{content:[{type:'tool_result',tool_use_id:'structured-call',content:'Structured output provided successfully'}]}}; + if(text==='hold'){ + const interrupted=new Promise(resolve=>{interrupt=resolve}); + process.send({type:'candidate_ready'}); + await Promise.race([interrupted,exited]); + } + } if(process.argv[1]==='classified'){ yield {type:'assistant',uuid:'assistant-error',session_id:'native',user_message_uuids:[user.uuid],parent_tool_use_id:null,error:'authentication_failed',message:{content:[]}}; yield {type:'result',uuid:'error-result',session_id:'native',user_message_uuids:[user.uuid],subtype:'error_during_execution',is_error:true,usage:{input_tokens:1,output_tokens:1},modelUsage:{},total_cost_usd:0}; continue; } - if(text==='hold') await Promise.race([new Promise(resolve=>{interrupt=resolve}),exited]); + if(text==='hold' && !options.outputFormat) await Promise.race([new Promise(resolve=>{interrupt=resolve}),exited]); if(options.abortController.signal.aborted)return; - yield {type:'result',uuid:'result-'+number,session_id:'native',user_message_uuids:[user.uuid],subtype:'success',is_error:false,result:'answer-'+number,usage:{input_tokens:1,output_tokens:1},modelUsage:{fixture:{inputTokens:number,outputTokens:number,costUSD:number/100}},total_cost_usd:number/100}; + yield {type:'result',uuid:'result-'+number,session_id:'native',user_message_uuids:[user.uuid],subtype:'success',is_error:false,result:'answer-'+number,...(options.outputFormat ? {structured_output:{number:9007199254740992}} : {}),usage:{input_tokens:1,output_tokens:1},modelUsage:{fixture:{inputTokens:number,outputTokens:number,costUSD:number/100}},total_cost_usd:number/100}; interrupt=undefined; yield {type:'command_lifecycle',uuid:user.uuid,state:'completed'}; } @@ -78,6 +96,7 @@ async function launch(t,mode="normal") { const send=value=>child.stdin.write(JSON.stringify(value)+"\n"); const start=(id,text)=>send({type:"turn_start",turn_id:id,input:[{content:[{type:"input_text",text}]}]}); send({type:"executor_prepare",preparation_deadline:Date.now()+60000,cwd:"/tmp",model:"fixture",system_prompt:"", + ...(mode==="structured" ? {output_format:{type:"json_schema",schema:{type:"object",properties:{number:{type:"integer"}}}}} : {}), ...(mode==="features" || mode.startsWith("pending-function") ? {functions:[{name:"lookup",description:"lookup",parameters:{type:"object",properties:{text:{type:"string"}}}}]} : {})}); await wait(()=>events.some(event=>event.type===(mode==="late-ready"?"error":"executor_ready"))); assert.equal(observations.filter(event=>event.type==="input").length,0); @@ -201,3 +220,27 @@ test("an initialization finishing past the original deadline never publishes rea assert.equal(observations.some(event=>event.type==="input"),false); assert.equal(observations.filter(event=>event.type==="native_closed").length,1); }); + +test("cancelled structured candidates stay private and a reused Turn preserves raw output and prose identities", {timeout:10000}, async t => { + const {child,events,observations,closed,wait,send,start}=await launch(t,"structured"); + start("first","hold"); + await wait(()=>observations.some(event=>event.type==="candidate_ready")); + send({type:"turn_cancel",turn_id:"first"}); + await wait(()=>events.some(event=>event.type==="turn_settled"&&event.turn_id==="first")); + assert.equal(events.some(event=>event.turn_id==="first"&&event.message?.phase==="final_answer"),false); + assert.ok(events.some(event=>event.turn_id==="first"&&event.type==="error"&&event.code==="cancelled")); + start("second","answer"); + await wait(()=>events.some(event=>event.type==="turn_settled"&&event.turn_id==="second")); + const own=events.filter(event=>event.turn_id==="second"); + assert.deepEqual(own.filter(event=>event.type==="output_message"&&event.message.status==="completed").map(event=>event.message),[ + {id:"structured-message",status:"completed",text:"ordinary prose"}, + {id:"structured-call",status:"completed",phase:"final_answer",text:'{"number":9007199254740993}'} + ]); + assert.ok(own.findIndex(event=>event.message?.phase==="final_answer") < own.findIndex(event=>event.type==="result")); + for(const turn of ["first","second"]){ + const settled=events.find(event=>event.turn_id===turn&&event.type==="turn_settled"); + assert.equal(settled.confirmed,true); + assert.equal(settled.reusable,true); + } + child.stdin.end();assert.deepEqual(await closed,{code:0,signal:null}); +}); diff --git a/packages/claude-sdk-adapter/tests/structured_output.test.mjs b/packages/claude-sdk-adapter/tests/structured_output.test.mjs index e15f2743c..8247c7d19 100644 --- a/packages/claude-sdk-adapter/tests/structured_output.test.mjs +++ b/packages/claude-sdk-adapter/tests/structured_output.test.mjs @@ -2,25 +2,114 @@ import test from "node:test"; import assert from "node:assert/strict"; import { StructuredOutput } from "../dist/structured_output.js"; import { parseStart } from "../dist/request.js"; -const call = (id, extra={}) => ({type:'assistant', session_id:'session', parent_tool_use_id:null, message:{content:[{type:'tool_use',id,name:'StructuredOutput',input:{}}]},...extra}); -const receipt = (id,error=false) => ({type:'user',session_id:'session',parent_tool_use_id:null,message:{content:[{type:'tool_result',tool_use_id:id,is_error:error}]}}); -const result = {type:'result',subtype:'success',is_error:false,structured_output:{number:9007199254740992},result:'{"number":9007199254740993}'}; -test('only the native confirmed terminal output is published, without JSON reserialization',()=>{ - const o=new StructuredOutput(); - o.consume(call('retry'),'session');o.consume(receipt('retry',true),'session'); - o.consume(call('final'),'session');assert.throws(()=>o.complete(result)); - o.consume(receipt('final'),'session'); - assert.deepEqual(o.complete(result),{type:'output_message',message:{id:'final',status:'completed',phase:'final_answer',text:result.result}}); - assert.throws(()=>o.complete(result)); + +const stream = event => ({type:"stream_event", uuid:crypto.randomUUID(), session_id:"session", parent_tool_use_id:null, event}); +const receipt = (id, error=false) => ({type:"user", uuid:`receipt-${id}`, session_id:"session", parent_tool_use_id:null, + message:{role:"user", content:[{type:"tool_result", tool_use_id:id, is_error:error, content:"Structured output provided successfully"}]}}); +const result = (raw, extra={}) => ({type:"result", uuid:"result", session_id:"session", subtype:"success", is_error:false, + structured_output:JSON.parse(raw), result:'{"memory":"unrelated summary"}', ...extra}); +const candidate = (id, raw, error=false) => [ + stream({type:"message_start", message:{id:`message-${id}`}}), + stream({type:"content_block_start", index:1, content_block:{type:"tool_use", id, name:"StructuredOutput", input:{}}}), + ...[raw.slice(0, -2), raw.slice(-2)].map(partial_json => stream({type:"content_block_delta", index:1, delta:{type:"input_json_delta", partial_json}})), + {type:"assistant", uuid:`snapshot-${id}`, session_id:"session", parent_tool_use_id:null, + message:{id:`message-${id}`, role:"assistant", content:[{type:"tool_use", id, name:"StructuredOutput", input:JSON.parse(raw)}]}}, + stream({type:"content_block_stop", index:1}), + stream({type:"message_stop"}), receipt(id, error), +]; +const consume = (observer, events) => events.forEach(event => observer.consume(event, "session")); + +test("the acknowledged raw candidate wins over unrelated generic result JSON", () => { + const raw = '{ "memory": "0123456789abcdef" }'; + const observer = new StructuredOutput(); + consume(observer, candidate("final", raw)); + assert.deepEqual(observer.complete(result(raw)), {type:"output_message", message:{id:"final", status:"completed", phase:"final_answer", text:raw}}); + assert.throws(() => observer.complete(result(raw))); }); -test('unrelated or failed results cannot publish a candidate',()=>{ - for(const extra of [{isReplay:true},{isSynthetic:true},{parent_tool_use_id:'child'},{session_id:'other'}]){ - const o=new StructuredOutput();o.consume(call('id',extra),'session');o.consume(receipt('id'),'session');assert.throws(()=>o.complete(result)); - } - for(const r of [{...result,subtype:'error_max_structured_output_retries'},{...result,structured_output:undefined},{...result,is_error:true}]){ - const o=new StructuredOutput();o.consume(call('id'),'session');o.consume(receipt('id'),'session');assert.throws(()=>o.complete(r)); - } + +test("failed retries never publish; raw integer digits survive native binary64 validation", () => { + const raw = '{"number":9007199254740993}'; + // Native equality is binary64: this proves byte fidelity, not exact-number schema validation. + assert.equal(JSON.parse(raw).number, 9007199254740992); + const observer = new StructuredOutput(); + consume(observer, candidate("retry", '{"number":0}', true)); + consume(observer, candidate("final", raw).slice(0, -1)); + assert.throws(() => observer.complete(result(raw))); + observer.consume(receipt("final"), "session"); + assert.equal(observer.complete(result(raw)).message.text, raw); }); + +test("streamed tool execution may acknowledge a snapshot before its block and message stop", () => { + const raw = '{"memory":"value"}'; + const events = candidate("final", raw); + const observer = new StructuredOutput(); + consume(observer, [...events.slice(0, 5), events[7]]); + assert.throws(() => observer.complete(result(raw))); + consume(observer, events.slice(5, 7)); + assert.equal(observer.complete(result(raw)).message.text, raw); +}); + +test("only live root events can confirm a candidate", () => { + const raw = '{"memory":"value"}'; + for (const extra of [{isReplay:true}, {isSynthetic:true}, {parent_tool_use_id:"child"}, {session_id:"other"}]) { + for (const index of [0, 1, 2, 4, 5, 6, 7]) { + const events = candidate("final", raw); + events[index] = {...events[index], ...extra}; + const observer = new StructuredOutput(); + assert.throws(() => { consume(observer, events); observer.complete(result(raw)); }); + } + } +}); + +test("incomplete, mismatched, duplicate and retracted native identities cannot publish", () => { + const raw = '{"memory":"value"}'; + const mutations = [ + events => events.filter((_, i) => i !== 1), + events => events.filter((_, i) => i !== 2), + events => events.filter((_, i) => i !== 4), + events => events.filter((_, i) => i !== 5), + events => events.filter((_, i) => i !== 6), + events => events.filter((_, i) => i !== 7), + events => { events[2].event.index = 2; return events; }, + events => { events[4].message.id = "other"; return events; }, + events => { events[4].message.content[0].id = "other"; return events; }, + events => { events[4].message.content[0].input = {memory:"other"}; return events; }, + events => { events[4].aborted = true; return events; }, + events => { events[4].error = "server_error"; return events; }, + events => [...events.slice(0, 2), events[1], ...events.slice(2)], + events => [...events.slice(0, 5), events[4], ...events.slice(5)], + events => { events[3].uuid = events[2].uuid; return events; }, + events => [...events, events[7]], + events => [...events, {...events[4], supersedes:["snapshot-final"]}], + events => [...events, {type:"system", subtype:"model_refusal_fallback", session_id:"session", retracted_message_uuids:["receipt-final"]}], + ]; + for (const mutate of mutations) { + const observer = new StructuredOutput(); + assert.throws(() => { consume(observer, mutate(candidate("final", raw))); observer.complete(result(raw)); }); + } +}); + +test("failed, ambiguous and mismatched final results cannot publish", () => { + const raw = '{"memory":"value"}'; + for (const extra of [{subtype:"error_max_structured_output_retries"}, {structured_output:undefined}, {structured_output:{memory:"other"}}, {is_error:true}, {session_id:"other"}]) { + const observer = new StructuredOutput(); + consume(observer, candidate("final", raw)); + assert.throws(() => observer.complete(result(raw, extra))); + } + for (const error of [false, true]) { + const observer = new StructuredOutput(); + assert.throws(() => { consume(observer, [...candidate("first", raw), ...candidate("second", raw, error)]); observer.complete(result(raw)); }); + } + const ambiguous = new StructuredOutput(); + consume(ambiguous, [...candidate("first", raw).slice(0, -1), ...candidate("second", raw).slice(0, -1), receipt("first"), receipt("second")]); + assert.throws(() => ambiguous.complete(result(raw))); + const failed = new StructuredOutput(); + consume(failed, candidate("failed", raw, true)); + assert.throws(() => failed.complete(result(raw))); + const observer = new StructuredOutput(); + assert.throws(() => { consume(observer, [...candidate("same", raw, true), ...candidate("same", raw)]); observer.complete(result(raw)); }); +}); + test('output configuration is a json_schema format',()=>{ const r={type:'start',input: [{ content: [{ type: "input_text", text: 'hello' }] }],model:'model',system_prompt:'',cwd:'/tmp',output_format:{type:'json_schema',schema:{type:'object'}}}; assert.deepEqual(parseStart(JSON.stringify(r)),r); From 10ed17fe9bd87b32905f0ebe82dc9ab188faa52f Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Fri, 9 Oct 2026 02:32:45 +0000 Subject: [PATCH 2/4] preserve native structured retries and signed zero --- .../src/structured_output.ts | 13 ++++++--- .../tests/structured_output.test.mjs | 27 +++++++++++++++++-- 2 files changed, 34 insertions(+), 6 deletions(-) diff --git a/packages/claude-sdk-adapter/src/structured_output.ts b/packages/claude-sdk-adapter/src/structured_output.ts index bf31429f6..ba0fa7d9f 100644 --- a/packages/claude-sdk-adapter/src/structured_output.ts +++ b/packages/claude-sdk-adapter/src/structured_output.ts @@ -2,7 +2,7 @@ import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk"; import { isDeepStrictEqual } from "node:util"; import type { MessageEvent } from "./messages.js"; -type Candidate = { id: string; text: string; snapshot?: string; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "published" }; +type Candidate = { id: string; text: string; snapshot?: string; input?: unknown; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "published" }; // StructuredOutput is the native terminal tool. Its acknowledged tool-use identity // owns the final JSON message; the parent assistant may already own ordinary prose. @@ -59,9 +59,9 @@ export class StructuredOutput { if (block.type !== "tool_use" || block.name !== "StructuredOutput") continue; const call = this.calls.get(block.id); if (!call || !this.active || message.message.id !== this.active.id || - ![...this.active.blocks.values()].includes(call) || call.snapshot || !message.uuid || message.error || message.aborted || - !isDeepStrictEqual(JSON.parse(call.text), block.input)) throw new Error("unmatched structured output snapshot"); + ![...this.active.blocks.values()].includes(call) || call.snapshot || !message.uuid || message.error || message.aborted) throw new Error("unmatched structured output snapshot"); call.snapshot = message.uuid; + call.input = block.input; } } else if (message.type === "user" && Array.isArray(message.message.content)) { for (const block of message.message.content) { @@ -69,6 +69,11 @@ export class StructuredOutput { const call = this.calls.get(block.tool_use_id); if (!call) continue; if (!call.snapshot || call.receipt || !message.uuid) throw new Error("invalid structured output receipt"); + // Failed native attempts may contain malformed JSON and must reach the + // SDK retry loop. Successful inputs must match JSON transport's zero semantics. + if (!block.is_error && !isDeepStrictEqual(JSON.parse(call.text, (_, value) => value === 0 ? 0 : value), call.input)) { + throw new Error("unmatched structured output input"); + } call.receipt = message.uuid; call.status = block.is_error ? "failed" : "accepted"; } @@ -81,7 +86,7 @@ export class StructuredOutput { if (message.type !== "result" || message.subtype !== "success" || message.is_error || message.session_id !== this.session || message.structured_output === undefined || this.active?.blocks.size || calls.some(call => !call.stopped || !call.receipt) || accepted.length !== 1 || - !isDeepStrictEqual(JSON.parse(accepted[0]!.text), message.structured_output)) throw new Error("unconfirmed structured output"); + !isDeepStrictEqual(accepted[0]!.input, message.structured_output)) throw new Error("unconfirmed structured output"); // Native validation compares binary64 values. Publish only the attributed raw // tool input: reserializing the validated object would lose original digits. const { id, text } = accepted[0]!; diff --git a/packages/claude-sdk-adapter/tests/structured_output.test.mjs b/packages/claude-sdk-adapter/tests/structured_output.test.mjs index 8247c7d19..b4dc0a31e 100644 --- a/packages/claude-sdk-adapter/tests/structured_output.test.mjs +++ b/packages/claude-sdk-adapter/tests/structured_output.test.mjs @@ -8,12 +8,12 @@ const receipt = (id, error=false) => ({type:"user", uuid:`receipt-${id}`, sessio message:{role:"user", content:[{type:"tool_result", tool_use_id:id, is_error:error, content:"Structured output provided successfully"}]}}); const result = (raw, extra={}) => ({type:"result", uuid:"result", session_id:"session", subtype:"success", is_error:false, structured_output:JSON.parse(raw), result:'{"memory":"unrelated summary"}', ...extra}); -const candidate = (id, raw, error=false) => [ +const candidate = (id, raw, error=false, input=JSON.parse(raw)) => [ stream({type:"message_start", message:{id:`message-${id}`}}), stream({type:"content_block_start", index:1, content_block:{type:"tool_use", id, name:"StructuredOutput", input:{}}}), ...[raw.slice(0, -2), raw.slice(-2)].map(partial_json => stream({type:"content_block_delta", index:1, delta:{type:"input_json_delta", partial_json}})), {type:"assistant", uuid:`snapshot-${id}`, session_id:"session", parent_tool_use_id:null, - message:{id:`message-${id}`, role:"assistant", content:[{type:"tool_use", id, name:"StructuredOutput", input:JSON.parse(raw)}]}}, + message:{id:`message-${id}`, role:"assistant", content:[{type:"tool_use", id, name:"StructuredOutput", input}]}}, stream({type:"content_block_stop", index:1}), stream({type:"message_stop"}), receipt(id, error), ]; @@ -49,6 +49,29 @@ test("streamed tool execution may acknowledge a snapshot before its block and me assert.equal(observer.complete(result(raw)).message.text, raw); }); +test("native malformed-input failure settles before a valid retry publishes", () => { + const raw = '{"number":'; + const input = {__unparsedToolInput:{raw, len:raw.length}}; + const observer = new StructuredOutput(); + consume(observer, candidate("malformed", raw, true, input)); + consume(observer, candidate("final", '{"number":7}')); + assert.equal(observer.complete(result('{"number":7}')).message.text, '{"number":7}'); + const invalid = new StructuredOutput(); + assert.throws(() => consume(invalid, candidate("malformed", raw, false, input))); +}); + +test("JSON transport normalizes negative zero without changing acknowledged raw text", () => { + const raw = '{"number":-0,"nested":[-0,{"zero":-0}]}'; + const transported = JSON.parse(JSON.stringify(JSON.parse(raw))); + const observer = new StructuredOutput(); + consume(observer, candidate("final", raw, false, transported)); + assert.equal(observer.complete(result(raw, {structured_output:transported})).message.text, raw); + // JSON transport turns infinity into null; it must not be treated like zero. + const overflowing = '{"number":1e999}'; + const invalid = new StructuredOutput(); + assert.throws(() => consume(invalid, candidate("overflow", overflowing, false, JSON.parse(JSON.stringify(JSON.parse(overflowing)))))); +}); + test("only live root events can confirm a candidate", () => { const raw = '{"memory":"value"}'; for (const extra of [{isReplay:true}, {isSynthetic:true}, {parent_tool_use_id:"child"}, {session_id:"other"}]) { From 38181a064b02522ea36eb1552eb1d7f60e8ebc5e Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Fri, 9 Oct 2026 02:50:25 +0000 Subject: [PATCH 3/4] preserve native structured candidate lifecycle --- contracts/agents-api/execution-tools.md | 2 +- contracts/agents-api/index.md | 2 +- contracts/agents-api/zh/execution-tools.md | 4 +- contracts/agents-api/zh/index.md | 4 +- .../src/structured_output.ts | 30 +++++---- .../tests/structured_output.test.mjs | 65 +++++++++++++++++-- 6 files changed, 81 insertions(+), 26 deletions(-) diff --git a/contracts/agents-api/execution-tools.md b/contracts/agents-api/execution-tools.md index 6ddfd2391..8ab01e5fe 100644 --- a/contracts/agents-api/execution-tools.md +++ b/contracts/agents-api/execution-tools.md @@ -41,7 +41,7 @@ The Worker fails previously claimed work without replay after execution loss; qu Core carries the schema in `ExecutionControls.OutputFormat` and requires `structured_output` in the Harness's declaration and the Runtime's heartbeat, only for requests that use the option. Frozen schemas reach preparation before input and apply to initial and resumed execution; Start cannot replace them. -The Claude adapter passes `outputFormat` to the pinned SDK and allows its native `StructuredOutput` terminal tool, which is internal and never an extra caller function. The adapter accumulates the original `input_json_delta` text by live root Session, native message ID, content-block index and tool-use ID. Only a completed matching assistant tool call, its successful receipt and a unique candidate consistent with the attributed successful SDK `structured_output` confirm that text. The adapter publishes it unchanged as a completed `final_answer` message with the native tool-use ID; parent assistant prose keeps its own ID. Unvalidated retries, missing or ambiguous identities, retracted work and cancelled candidates never become the answer. The generic SDK `result` text is not the structured payload, and the adapter never serializes `structured_output` back to JSON. Raw numeric digits are preserved; exact-number schema validation remains a [known gap](./index.md#known-gaps). The stream follows the official message sequence with the whole text in one `output_text.delta`. The bridge advertises the operation only when it reports `structured_output`, and a workspace Runtime also needs `workspace_structured_output`. +The Claude adapter passes `outputFormat` to the pinned SDK and allows its native `StructuredOutput` terminal tool, which is internal and never an extra caller function. The adapter accumulates the original `input_json_delta` text by live root Session, native message ID, content-block index and tool-use ID. Only a completed matching assistant tool call, its successful receipt and a unique candidate consistent with the attributed successful SDK `structured_output` confirm that text. The adapter publishes it unchanged as a completed `final_answer` message with the native tool-use ID; parent assistant prose keeps its own ID. Native retry and fallback decisions stay in the SDK. Closed partial inputs without a completed tool snapshot and explicitly retracted candidates are discarded; replacements must establish the same confirmation chain. Missing or ambiguous identities and cancelled candidates never become the answer. The generic SDK `result` text is not the structured payload, and the adapter never serializes `structured_output` back to JSON. Raw numeric digits are preserved; exact-number schema validation and native fallbacks without raw tool-input events remain [known gaps](./index.md#known-gaps). The stream follows the official message sequence with the whole text in one `output_text.delta`. The bridge advertises the operation only when it reports `structured_output`, and a workspace Runtime also needs `workspace_structured_output`. ## Deferred function discovery diff --git a/contracts/agents-api/index.md b/contracts/agents-api/index.md index f862272bd..ea5548ca4 100644 --- a/contracts/agents-api/index.md +++ b/contracts/agents-api/index.md @@ -110,7 +110,7 @@ Each item is Core's deliberate or native behavior where the official service beh - Explicit reasoning effort or summary, service tiers other than `auto`, enabled `web_search` and enabled programmatic tool calling are saved but rejected at Session admission. - Harness support differs as each [declaration](./harness-onboarding.md#declare-support) states. Codex has no structured output or `tool_search`. Claude Code takes no whitespace-only text, `medium` verbosity only and function-result images only inline and in successful results; it rejects Subagents with MCP, including MCP servers that Environment Plugins install; structured output with Subagents, MCP, installed capabilities or `tool_search`; and `tool_search` with MCP or installed capabilities. MiniMax Code has no public functions, no service-origin MCP, no image input, no whitespace-only text, no required MCP and `medium` verbosity only, and takes `allowed_tools` null only. Each Harness reserves an MCP `server_label`: `codex_apps` for Codex, `functions` for Claude Code and `oac_workspace` for MiniMax Code. Claude Code also requires labels to match `^[a-zA-Z0-9_-]+$` and `allowed_tools` names to match `^[a-zA-Z0-9_.-]+$`. -- Claude structured output preserves the [acknowledged raw tool input](./execution-tools.md#structured-output), while native validation and parsed-value correlation use binary64. Distinct raw numbers can compare equal after parsing, so this does not establish exact-number schema conformance, even when every schema literal passes admission. Exact numeric output validation remains unqualified. +- Claude structured output preserves the [acknowledged raw tool input](./execution-tools.md#structured-output), while native validation and parsed-value correlation use binary64. Distinct raw numbers can compare equal after parsing, so this does not establish exact-number schema conformance, even when every schema literal passes admission. Exact numeric output validation remains unqualified. A native streaming-to-nonstreaming fallback can emit an assistant tool snapshot without raw tool-input events. The adapter rejects such output because the original JSON text is unavailable; parsed input and generic result text cannot substitute for it. - Model-derived reasoning defaults are not resolved. - MCP tools support the `http` transport only; `stdio` is rejected, and so is an inline `authorization` on a Session MCP transport ([HTTP MCP](./execution-tools.md#http-mcp)). diff --git a/contracts/agents-api/zh/execution-tools.md b/contracts/agents-api/zh/execution-tools.md index 2f31fe5cc..f5a388ce1 100644 --- a/contracts/agents-api/zh/execution-tools.md +++ b/contracts/agents-api/zh/execution-tools.md @@ -1,7 +1,7 @@ --- title: "执行工具" source: contracts/agents-api/execution-tools.md -source_hash: fbd6a39ac815c0e3af66f72d3690c8ab7b705f45a8f5eedbdf6c00a33eebe355 +source_hash: 93a806df9242467c895429ebdec03e3daf28dd5f652f1f29bbda4dcfc18995f0 --- Agent 在 `tools` 中声明应用函数、控制项和 MCP 服务器,并可在 `text.format` 中声明输出 schema。本契约说明 Core 如何验证声明、哪些内容跨越 Runtime 边界,以及调用方如何恢复待执行操作。每个 Harness 的[声明](harness-onboarding.md#declare-support)说明它支持其中哪些内容。原生工作区工具和 Environment Plugin MCP 属于 [Environment](environments.md#skills-plugins-and-environment-mcp)。 @@ -43,7 +43,7 @@ SSE 仅提供实时事件。重启或流丢失后,读取 Session 的 `required Core 在 `ExecutionControls.OutputFormat` 中携带 schema,仅对使用该选项的请求要求 Harness 的声明和 Runtime 的心跳都支持 `structured_output`。冻结的 schema 在输入之前送达准备阶段,适用于初次和恢复执行;Start 不能替换它。 -Claude 适配器将 `outputFormat` 传给固定版本 SDK,并允许原生 `StructuredOutput` 终态工具;该工具属于内部,不是额外的调用方函数。适配器按实时根 Session、原生消息 ID、内容块索引和 tool-use ID 累积原始 `input_json_delta` 文本。只有匹配的完整 assistant 工具调用、成功回执,以及与已归属的成功 SDK `structured_output` 一致的唯一候选,才能确认该文本。适配器将其原样发布为已完成的 `final_answer` 消息,使用原生 tool-use ID;父 assistant 文本保留自己的 ID。未验证重试、缺失或歧义身份、已撤回工作和已取消候选不会成为答案。通用 SDK `result` 文本不是结构化载荷,适配器不会把 `structured_output` 重新序列化为 JSON。原始数字文本会被保留;精确数值 schema 验证仍是[已知缺口](./index.md#known-gaps)。流遵循官方消息顺序,将整段文本放入一个 `output_text.delta`。桥接层仅在报告 `structured_output` 时声明该操作,工作区 Runtime 还需要 `workspace_structured_output`。 +Claude 适配器将 `outputFormat` 传给固定版本 SDK,并允许原生 `StructuredOutput` 终态工具;该工具属于内部,不是额外的调用方函数。适配器按实时根 Session、原生消息 ID、内容块索引和 tool-use ID 累积原始 `input_json_delta` 文本。只有匹配的完整 assistant 工具调用、成功回执,以及与已归属的成功 SDK `structured_output` 一致的唯一候选,才能确认该文本。适配器将其原样发布为已完成的 `final_answer` 消息,使用原生 tool-use ID;父 assistant 文本保留自己的 ID。原生重试和 fallback 决策仍由 SDK 负责。已关闭但没有完整工具 snapshot 的部分输入及显式撤回的候选会被弃置;替代候选必须建立相同的确认链。缺失或歧义身份以及已取消候选不会成为答案。通用 SDK `result` 文本不是结构化载荷,适配器不会把 `structured_output` 重新序列化为 JSON。原始数字文本会被保留;精确数值 schema 验证和不提供原始工具输入事件的原生 fallback 仍是[已知缺口](./index.md#known-gaps)。流遵循官方消息顺序,将整段文本放入一个 `output_text.delta`。桥接层仅在报告 `structured_output` 时声明该操作,工作区 Runtime 还需要 `workspace_structured_output`。 ## 延迟函数发现 {#deferred-function-discovery} diff --git a/contracts/agents-api/zh/index.md b/contracts/agents-api/zh/index.md index b35444d53..0531232af 100644 --- a/contracts/agents-api/zh/index.md +++ b/contracts/agents-api/zh/index.md @@ -1,7 +1,7 @@ --- title: "Agents API 覆盖台账" source: contracts/agents-api/index.md -source_hash: 34f04869ed4a0cf01754d7375688c7176083eff5ec26f68321adbace76158be9 +source_hash: 4bbbd2097e76481ec50fc7bc41de7855e88fb8346e65f389ec0230f5654f15ce --- Core 旨在以下方固定版本为准支持完整的 OpenAI Agents API([public API rule](https://github.com/MiniMax-AI/OpenAgentCore/blob/main/AGENTS.md#public-api))。本台账记录 Core 对各项资源实现了哪些内容、哪些契约保存其详细信息,并列出相对于 OpenAI 服务的所有已知差异和所有未解决缺口。[API namespaces and credentials](../../../docs/zh/api/index.md) 说明谁调用哪些 API;[Agents API guide](../../../docs/zh/api/public-agent-api.md) 介绍使用方法。 @@ -112,7 +112,7 @@ Core 自身字段位于 `x_agents_core` 中([Core extensions](../../../docs/zh - 显式指定推理强度或摘要、使用 `auto` 之外的服务层级、启用 `web_search` 或启用程序化工具调用,这些设置都会被保存,但在 Session 准入时会被拒绝。 - 各 Harness 的支持差异以其[声明](harness-onboarding.md#declare-support)为准。Codex 不支持结构化输出或 `tool_search`。Claude Code 不接受仅含空白的文本,只支持 `medium` 详细程度,函数结果图像只能内联且只能出现在成功结果中;它拒绝子智能体与 MCP(包括 Environment Plugins 安装的 MCP server)同时使用,拒绝结构化输出与子智能体、MCP、已安装能力或 `tool_search` 同时使用,也拒绝 `tool_search` 与 MCP 或已安装能力同时使用。MiniMax Code 不提供公共 functions,没有服务源 MCP,不支持图像输入、仅含空白的文本和必需 MCP,只支持 `medium` 详细程度,且只接受值为 null 的 `allowed_tools`。每个 Harness 都保留一个 MCP `server_label`:Codex 保留 `codex_apps`,Claude Code 保留 `functions`,MiniMax Code 保留 `oac_workspace`。Claude Code 还要求标签匹配 `^[a-zA-Z0-9_-]+$`,`allowed_tools` 中的名称匹配 `^[a-zA-Z0-9_.-]+$`。 -- Claude 结构化输出保留[已确认的原始工具输入](./execution-tools.md#structured-output),但原生验证和解析值关联使用 binary64。不同的原始数字在解析后可能相等,因此即使 schema 的所有数值字面量都通过准入,也不能据此确认精确数值的 schema 符合性。精确数值输出验证仍未经资格验证。 +- Claude 结构化输出保留[已确认的原始工具输入](./execution-tools.md#structured-output),但原生验证和解析值关联使用 binary64。不同的原始数字在解析后可能相等,因此即使 schema 的所有数值字面量都通过准入,也不能据此确认精确数值的 schema 符合性。精确数值输出验证仍未经资格验证。 原生从流式到非流式的 fallback 可能只产生 assistant 工具 snapshot,而没有原始工具输入事件。由于原始 JSON 文本不可用,适配器会拒绝此类输出;解析后的输入和通用结果文本不能替代它。 - 由模型推导出的推理默认值不会被解析确定。 - MCP 工具仅支持 `http` 传输,`stdio` 会被拒绝,Session MCP 传输中的内联 `authorization` 也会被拒绝([HTTP MCP](execution-tools.md#http-mcp))。 diff --git a/packages/claude-sdk-adapter/src/structured_output.ts b/packages/claude-sdk-adapter/src/structured_output.ts index ba0fa7d9f..86679b3cf 100644 --- a/packages/claude-sdk-adapter/src/structured_output.ts +++ b/packages/claude-sdk-adapter/src/structured_output.ts @@ -2,13 +2,12 @@ import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk"; import { isDeepStrictEqual } from "node:util"; import type { MessageEvent } from "./messages.js"; -type Candidate = { id: string; text: string; snapshot?: string; input?: unknown; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "published" }; +type Candidate = { id: string; text: string; snapshot?: string; input?: unknown; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "published" | "discarded" }; // StructuredOutput is the native terminal tool. Its acknowledged tool-use identity // owns the final JSON message; the parent assistant may already own ordinary prose. export class StructuredOutput { private readonly calls = new Map(); - private readonly messages = new Set(); private readonly events = new Set(); private active?: { id: string; blocks: Map }; private session = ""; @@ -18,8 +17,8 @@ export class StructuredOutput { ("isReplay" in message && message.isReplay) || ("isSynthetic" in message && message.isSynthetic)) return; const retracted = message.type === "assistant" ? message.supersedes : message.type === "system" && message.subtype === "model_refusal_fallback" ? message.retracted_message_uuids : undefined; - if (retracted?.some(id => [...this.calls.values()].some(call => call.snapshot === id || call.receipt === id))) { - throw new Error("retracted structured output"); + if (retracted) for (const call of this.calls.values()) { + if (retracted.some(id => call.snapshot === id || call.receipt === id)) call.status = "discarded"; } if (!("parent_tool_use_id" in message) || message.parent_tool_use_id !== null) return; this.session = session; @@ -28,13 +27,14 @@ export class StructuredOutput { this.events.add(message.uuid); const event = message.event; if (event.type === "message_start") { - if (!event.message.id || this.messages.has(event.message.id) || this.active?.blocks.size) throw new Error("invalid structured message identity"); - this.messages.add(event.message.id); + if (!event.message.id || this.active?.blocks.size) throw new Error("invalid structured message identity"); this.active = { id: event.message.id, blocks: new Map() }; } else if (event.type === "content_block_start" && event.content_block.type === "tool_use" && event.content_block.name === "StructuredOutput") { const id = event.content_block.id; - if (!this.active || !id || this.calls.has(id) || this.active.blocks.has(event.index) || - [...this.calls.values()].some(call => call.status === "accepted")) throw new Error("invalid structured output identity"); + // An abandoned prefix never acquired an executable native call identity. + const previous = this.calls.get(id); + if (!this.active || !id || this.active.blocks.has(event.index) || + previous && (previous.status !== "discarded" || previous.snapshot)) throw new Error("invalid structured output identity"); const call: Candidate = { id, text: "", stopped: false }; this.calls.set(id, call); this.active.blocks.set(event.index, call); @@ -47,8 +47,10 @@ export class StructuredOutput { } else if (event.type === "content_block_stop") { const call = this.active?.blocks.get(event.index); if (call) { - if (!call.snapshot || call.stopped) throw new Error("unconfirmed structured output block"); + if (call.stopped) throw new Error("duplicate structured output block stop"); call.stopped = true; + // Native connection retries close partial blocks without executing a tool. + if (!call.snapshot) call.status = "discarded"; } } else if (event.type === "message_stop") { if (this.active && [...this.active.blocks.values()].some(call => !call.stopped)) throw new Error("incomplete structured output message"); @@ -59,7 +61,7 @@ export class StructuredOutput { if (block.type !== "tool_use" || block.name !== "StructuredOutput") continue; const call = this.calls.get(block.id); if (!call || !this.active || message.message.id !== this.active.id || - ![...this.active.blocks.values()].includes(call) || call.snapshot || !message.uuid || message.error || message.aborted) throw new Error("unmatched structured output snapshot"); + ![...this.active.blocks.values()].includes(call) || call.snapshot || call.stopped || !message.uuid || message.error || message.aborted) throw new Error("unmatched structured output snapshot"); call.snapshot = message.uuid; call.input = block.input; } @@ -71,11 +73,11 @@ export class StructuredOutput { if (!call.snapshot || call.receipt || !message.uuid) throw new Error("invalid structured output receipt"); // Failed native attempts may contain malformed JSON and must reach the // SDK retry loop. Successful inputs must match JSON transport's zero semantics. - if (!block.is_error && !isDeepStrictEqual(JSON.parse(call.text, (_, value) => value === 0 ? 0 : value), call.input)) { + if (call.status !== "discarded" && !block.is_error && !isDeepStrictEqual(JSON.parse(call.text, (_, value) => value === 0 ? 0 : value), call.input)) { throw new Error("unmatched structured output input"); } call.receipt = message.uuid; - call.status = block.is_error ? "failed" : "accepted"; + if (call.status !== "discarded") call.status = block.is_error ? "failed" : "accepted"; } } } @@ -84,8 +86,8 @@ export class StructuredOutput { const calls = [...this.calls.values()]; const accepted = calls.filter(call => call.status === "accepted"); if (message.type !== "result" || message.subtype !== "success" || message.is_error || - message.session_id !== this.session || message.structured_output === undefined || this.active?.blocks.size || - calls.some(call => !call.stopped || !call.receipt) || accepted.length !== 1 || + message.session_id !== this.session || message.structured_output === undefined || + calls.some(call => call.status !== "discarded" && (!call.stopped || !call.receipt)) || accepted.length !== 1 || !isDeepStrictEqual(accepted[0]!.input, message.structured_output)) throw new Error("unconfirmed structured output"); // Native validation compares binary64 values. Publish only the attributed raw // tool input: reserializing the validated object would lose original digits. diff --git a/packages/claude-sdk-adapter/tests/structured_output.test.mjs b/packages/claude-sdk-adapter/tests/structured_output.test.mjs index b4dc0a31e..59a1e7e88 100644 --- a/packages/claude-sdk-adapter/tests/structured_output.test.mjs +++ b/packages/claude-sdk-adapter/tests/structured_output.test.mjs @@ -72,10 +72,68 @@ test("JSON transport normalizes negative zero without changing acknowledged raw assert.throws(() => consume(invalid, candidate("overflow", overflowing, false, JSON.parse(JSON.stringify(JSON.parse(overflowing)))))); }); +test("native connection retry abandons a closed partial before a confirmed replacement", () => { + const raw = '{"number":7}'; + const events = candidate("abandoned", raw); + // The native retry producer closes the partial without constructing a tool snapshot. + const partial = [...events.slice(0, 3), ...events.slice(5, 7)]; + const observer = new StructuredOutput(); + consume(observer, partial); + assert.throws(() => observer.complete(result(raw))); + // Providers may reuse IDs on retry; the discarded prefix never became a native call. + consume(observer, candidate("abandoned", raw)); + assert.equal(observer.complete(result(raw)).message.id, "abandoned"); + for (const late of [events[4], events[7]]) { + const invalid = new StructuredOutput(); + consume(invalid, partial); + assert.throws(() => invalid.consume(late, "session")); + } +}); + +test("native retraction permits replacement raw input before its supersedes snapshot", () => { + const raw = '{"number":7}'; + for (const lateReceipt of [false, true]) { + const observer = new StructuredOutput(); + const old = candidate("retracted", raw); + consume(observer, lateReceipt ? old.slice(0, -1) : old); + const replacement = candidate("replacement", raw); + replacement[4].supersedes = ["snapshot-retracted"]; + consume(observer, replacement); + if (lateReceipt) observer.consume(old[7], "session"); + // The final banner repeats the earlier supersedes notice idempotently. + observer.consume({type:"system", subtype:"model_refusal_fallback", session_id:"session", retracted_message_uuids:["snapshot-retracted","receipt-retracted"]}, "session"); + assert.equal(observer.complete(result(raw)).message.id, "replacement"); + } +}); + +test("a completed tool block survives native partial finalization without message_stop", () => { + const raw = '{"number":7}'; + const events = candidate("final", raw); + const observer = new StructuredOutput(); + consume(observer, [...events.slice(0, 6), {type:"assistant", session_id:"session", parent_tool_use_id:null, + error:"server_error", message:{content:[{type:"text", text:"Connection lost mid-response."}]}}, events[7]]); + assert.equal(observer.complete(result(raw)).message.text, raw); +}); + +test("retracted materialized tool identities cannot be reused or revived", () => { + const raw = '{"number":7}'; + const observer = new StructuredOutput(); + consume(observer, candidate("retracted", raw)); + observer.consume({type:"system", subtype:"model_refusal_fallback", session_id:"session", retracted_message_uuids:["snapshot-retracted"]}, "session"); + assert.throws(() => observer.complete(result(raw))); + assert.throws(() => consume(observer, candidate("retracted", raw))); +}); + +test("an assistant-only native fallback cannot supply original tool-input text", () => { + const raw = '{"number":7}'; + const observer = new StructuredOutput(); + assert.throws(() => consume(observer, candidate("fallback", raw).slice(4))); +}); + test("only live root events can confirm a candidate", () => { const raw = '{"memory":"value"}'; for (const extra of [{isReplay:true}, {isSynthetic:true}, {parent_tool_use_id:"child"}, {session_id:"other"}]) { - for (const index of [0, 1, 2, 4, 5, 6, 7]) { + for (const index of [0, 1, 2, 4, 5, 7]) { const events = candidate("final", raw); events[index] = {...events[index], ...extra}; const observer = new StructuredOutput(); @@ -91,7 +149,6 @@ test("incomplete, mismatched, duplicate and retracted native identities cannot p events => events.filter((_, i) => i !== 2), events => events.filter((_, i) => i !== 4), events => events.filter((_, i) => i !== 5), - events => events.filter((_, i) => i !== 6), events => events.filter((_, i) => i !== 7), events => { events[2].event.index = 2; return events; }, events => { events[4].message.id = "other"; return events; }, @@ -119,10 +176,6 @@ test("failed, ambiguous and mismatched final results cannot publish", () => { consume(observer, candidate("final", raw)); assert.throws(() => observer.complete(result(raw, extra))); } - for (const error of [false, true]) { - const observer = new StructuredOutput(); - assert.throws(() => { consume(observer, [...candidate("first", raw), ...candidate("second", raw, error)]); observer.complete(result(raw)); }); - } const ambiguous = new StructuredOutput(); consume(ambiguous, [...candidate("first", raw).slice(0, -1), ...candidate("second", raw).slice(0, -1), receipt("first"), receipt("second")]); assert.throws(() => ambiguous.complete(result(raw))); From 2b849caa17fe11d4ea43784b9c2c6f6e0d744f57 Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Fri, 9 Oct 2026 02:53:41 +0000 Subject: [PATCH 4/4] scope and match native structured candidates --- .../src/structured_output.ts | 19 +++++----- .../tests/structured_output.test.mjs | 36 +++++++++++++++++++ 2 files changed, 46 insertions(+), 9 deletions(-) diff --git a/packages/claude-sdk-adapter/src/structured_output.ts b/packages/claude-sdk-adapter/src/structured_output.ts index 86679b3cf..616816978 100644 --- a/packages/claude-sdk-adapter/src/structured_output.ts +++ b/packages/claude-sdk-adapter/src/structured_output.ts @@ -2,7 +2,7 @@ import type { SDKMessage } from "@anthropic-ai/claude-agent-sdk"; import { isDeepStrictEqual } from "node:util"; import type { MessageEvent } from "./messages.js"; -type Candidate = { id: string; text: string; snapshot?: string; input?: unknown; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "published" | "discarded" }; +type Candidate = { id: string; text: string; snapshot?: string; input?: unknown; receipt?: string; stopped: boolean; status?: "failed" | "accepted" | "settled" | "discarded" }; // StructuredOutput is the native terminal tool. Its acknowledged tool-use identity // owns the final JSON message; the parent assistant may already own ordinary prose. @@ -15,8 +15,8 @@ export class StructuredOutput { consume(message: SDKMessage, session: string): void { if (!session || message.session_id !== session || ("isReplay" in message && message.isReplay) || ("isSynthetic" in message && message.isSynthetic)) return; - const retracted = message.type === "assistant" ? message.supersedes : - message.type === "system" && message.subtype === "model_refusal_fallback" ? message.retracted_message_uuids : undefined; + const retracted = message.type === "assistant" && message.parent_tool_use_id === null ? message.supersedes : + message.type === "system" && message.subtype === "model_refusal_fallback" && message.scope !== "local" ? message.retracted_message_uuids : undefined; if (retracted) for (const call of this.calls.values()) { if (retracted.some(id => call.snapshot === id || call.receipt === id)) call.status = "discarded"; } @@ -83,16 +83,17 @@ export class StructuredOutput { } complete(message: SDKMessage): MessageEvent { - const calls = [...this.calls.values()]; - const accepted = calls.filter(call => call.status === "accepted"); if (message.type !== "result" || message.subtype !== "success" || message.is_error || - message.session_id !== this.session || message.structured_output === undefined || - calls.some(call => call.status !== "discarded" && (!call.stopped || !call.receipt)) || accepted.length !== 1 || - !isDeepStrictEqual(accepted[0]!.input, message.structured_output)) throw new Error("unconfirmed structured output"); + message.session_id !== this.session || message.structured_output === undefined) throw new Error("unconfirmed structured output"); + const calls = [...this.calls.values()]; + const accepted = calls.filter(call => call.status === "accepted" && isDeepStrictEqual(call.input, message.structured_output)); + if (calls.some(call => call.status !== "discarded" && (!call.stopped || !call.receipt)) || accepted.length !== 1) { + throw new Error("unconfirmed structured output"); + } // Native validation compares binary64 values. Publish only the attributed raw // tool input: reserializing the validated object would lose original digits. const { id, text } = accepted[0]!; - accepted[0]!.status = "published"; + for (const call of calls) if (call.status === "accepted") call.status = "settled"; return { type: "output_message", message: { id, status: "completed", phase: "final_answer", text } }; } } diff --git a/packages/claude-sdk-adapter/tests/structured_output.test.mjs b/packages/claude-sdk-adapter/tests/structured_output.test.mjs index 59a1e7e88..23588ee1c 100644 --- a/packages/claude-sdk-adapter/tests/structured_output.test.mjs +++ b/packages/claude-sdk-adapter/tests/structured_output.test.mjs @@ -130,6 +130,42 @@ test("an assistant-only native fallback cannot supply original tool-input text", assert.throws(() => consume(observer, candidate("fallback", raw).slice(4))); }); +test("child-scope retraction notices cannot revoke a root candidate", () => { + const raw = '{"number":7}'; + for (const notice of [ + {type:"assistant", parent_tool_use_id:"child", supersedes:["snapshot-final"], message:{content:[]}}, + {type:"system", subtype:"model_refusal_fallback", scope:"local", retracted_message_uuids:["snapshot-final"]}, + ]) { + const observer = new StructuredOutput(); + consume(observer, candidate("final", raw)); + observer.consume({...notice, session_id:"session"}, "session"); + assert.equal(observer.complete(result(raw)).message.id, "final"); + } +}); + +test("the final native value uniquely selects among successful tools in one message", () => { + for (const [firstRaw, finalRaw, ambiguous] of [ + ['{"number":1}', '{"number":2}', false], + ['{"number":9007199254740992}', '{"number":9007199254740993}', true], + ]) { + const first = candidate("first", firstRaw); + const last = candidate("last", finalRaw); + last[4].message.id = first[4].message.id; + for (const frame of last) if (frame.event && "index" in frame.event) frame.event.index = 2; + const observer = new StructuredOutput(); + consume(observer, [...first.slice(0, 6), ...last.slice(1, 7), first[7], last[7]]); + if (ambiguous) assert.throws(() => observer.complete(result(finalRaw))); + else { + assert.deepEqual(observer.complete(result(finalRaw)).message, + {id:"last", status:"completed", phase:"final_answer", text:finalRaw}); + // Steering can yield another native result within this public Turn. + assert.throws(() => observer.complete(result(firstRaw))); + consume(observer, candidate("next", firstRaw)); + assert.equal(observer.complete(result(firstRaw)).message.id, "next"); + } + } +}); + test("only live root events can confirm a candidate", () => { const raw = '{"memory":"value"}'; for (const extra of [{isReplay:true}, {isSynthetic:true}, {parent_tool_use_id:"child"}, {session_id:"other"}]) {