From 067fd74c352c989c293e44f406c22008ed17a025 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:04:01 +0900 Subject: [PATCH 01/24] fix(cursor): bound capability reads and buffered tool budgets (#5533) Carries #5533 (and the closed #5233 it consolidates) onto current dev. Co-authored-by: Epinephrine --- src/adapters/cursor/protobuf-events.ts | 52 +++++++++++---- src/integrations/cursor-effort-table.ts | 63 +++++++++++++++---- structure/clients/integrations.md | 8 +++ structure/providers/cursor.md | 5 +- .../cursor/cursor-effort-table.test.ts | 48 +++++++++++--- .../cursor/cursor-protobuf-events.test.ts | 28 ++++++++- 6 files changed, 168 insertions(+), 36 deletions(-) diff --git a/src/adapters/cursor/protobuf-events.ts b/src/adapters/cursor/protobuf-events.ts index 1bc771a0098..c10c3f20ad1 100644 --- a/src/adapters/cursor/protobuf-events.ts +++ b/src/adapters/cursor/protobuf-events.ts @@ -189,8 +189,8 @@ export interface CursorProtobufEventState { pendingTextToolCall?: string; /** Constant-space scanner used after an incomplete textual marker exceeds its retained byte cap. */ suppressedTextToolCall?: SuppressedTextToolCallScan; - /** Parsed textual fallback calls held until turn finalization establishes that no real frame won. */ - bufferedTextToolCalls?: DrainedTextToolCall[]; + /** Budgeted textual fallback calls held until turn finalization establishes that no real frame won. */ + bufferedTextToolCalls?: Array; /** True once this turn carries any real client-tool frame, including an incomplete one. */ sawRealClientToolCall?: boolean; /** Monotonic id suffix for tool calls promoted from text markers. */ @@ -1077,7 +1077,10 @@ export function mapSyntheticMcpExecToToolEvents( ): CursorServerMessage[] { if (args.providerIdentifier !== OCX_RESPONSES_TOOL_PROVIDER) return []; if (options.state?.terminated) return []; - if (options.state) options.state.sawRealClientToolCall = true; + if (options.state) { + discardBufferedTextToolCalls(options.state); + options.state.sawRealClientToolCall = true; + } if (options.allowEmptyArgs !== true && !hasMcpArgBytes(args)) return []; const cursorWireName = mcpWireNameFromArgs(args); if (!cursorWireName) return [{ type: "error", message: "Cursor requested a Responses tool without a tool name" }]; @@ -1148,10 +1151,16 @@ function recordToolCall(state: CursorProtobufEventState, callId: string, cursorW } function recordRealToolCall(state: CursorProtobufEventState, callId: string, cursorWireName: string): CursorServerMessage[] { + discardBufferedTextToolCalls(state); state.sawRealClientToolCall = true; return recordToolCall(state, callId, cursorWireName); } +function discardBufferedTextToolCalls(state: CursorProtobufEventState): void { + for (const call of state.bufferedTextToolCalls ?? []) state.translatorBudget?.closeCall(call.callId); + delete state.bufferedTextToolCalls; +} + /** * Emit a completed client tool call as one atomic unit: `tool_call_start` (deferred from open time), * the full normalized arguments delta when present, then `tool_call_end`. The call must already be @@ -1313,10 +1322,21 @@ export function mapCursorProtobufServerMessage( || !advertised || (state.bufferedTextToolCalls?.length ?? 0) >= state.maxClientToolCalls ) continue; - (state.bufferedTextToolCalls ??= []).push({ - name: advertised, - args: normalizeJsonText(call.args, advertised, state), - }); + const args = normalizeJsonText(call.args, advertised, state); + state.textToolCallSeq = (state.textToolCallSeq ?? 0) + 1; + const callId = `textcall_${state.textToolCallSeq}`; + state.translatorBudget?.openCall(callId); + try { + const reservation = state.translatorBudget?.reserveTransient( + Buffer.byteLength(args), + { kind: "tool_args", callId }, + ); + reservation?.commitRetained(); + (state.bufferedTextToolCalls ??= []).push({ name: advertised, args, callId }); + } catch (error) { + state.translatorBudget?.closeCall(callId); + throw error; + } } return out; } @@ -1346,7 +1366,10 @@ export function mapCursorProtobufServerMessage( const out: CursorServerMessage[] = []; if (state.completedToolCalls.has(update.value.callId)) return []; const name = mcpCursorWireName(update.value.toolCall); - if (name) state.sawRealClientToolCall = true; + if (name) { + discardBufferedTextToolCalls(state); + state.sawRealClientToolCall = true; + } const args = mcpArgsFromToolCall(update.value.toolCall); const openBeforeStart = state.openToolCalls.get(update.value.callId); // Empty-arg completion handling: @@ -1454,6 +1477,7 @@ export function finalizeTurnEvents(state: CursorProtobufEventState): CursorServe const bufferedTextToolCalls = state.bufferedTextToolCalls ?? []; delete state.bufferedTextToolCalls; if (state.openToolCalls.size > 0) { + for (const call of bufferedTextToolCalls) state.translatorBudget?.closeCall(call.callId); const openCallIds = [...state.openToolCalls.keys()]; const openIds = openCallIds.join(", "); // Clear so a second turnEnded (should not happen, but defensive) doesn't re-emit. @@ -1464,11 +1488,15 @@ export function finalizeTurnEvents(state: CursorProtobufEventState): CursorServe const out: CursorServerMessage[] = []; if (!state.sawRealClientToolCall) { for (const call of bufferedTextToolCalls) { - state.textToolCallSeq = (state.textToolCallSeq ?? 0) + 1; - const callId = `textcall_${state.textToolCallSeq}`; - out.push(...recordToolCall(state, callId, call.name)); - if (state.openToolCalls.has(callId)) out.push(...commitToolCall(state, callId, call.args)); + out.push(...recordToolCall(state, call.callId, call.name)); + const open = state.openToolCalls.get(call.callId); + if (open) { + open.args = call.args; + out.push(...commitToolCall(state, call.callId, call.args)); + } else state.translatorBudget?.closeCall(call.callId); } + } else { + for (const call of bufferedTextToolCalls) state.translatorBudget?.closeCall(call.callId); } // Surface the absolute context size (when Cursor reported a checkpoint) as both totalTokens and // the estimated input side of Codex's visible `input + output` counter. Codex status lines can diff --git a/src/integrations/cursor-effort-table.ts b/src/integrations/cursor-effort-table.ts index bca6a53c305..b69de7dfef6 100644 --- a/src/integrations/cursor-effort-table.ts +++ b/src/integrations/cursor-effort-table.ts @@ -8,7 +8,7 @@ * instead of a hand-copied mirror. Read-only, size-bounded, cached by (path, mtime, size); * any parse failure yields null so the caller falls back to the static mirror. */ -import { readFileSync, statSync } from "node:fs"; +import { closeSync, constants, fstatSync, openSync, readSync } from "node:fs"; import { join } from "node:path"; import type { CursorInstall } from "./cursor-detect"; @@ -111,32 +111,69 @@ function splitStrings(list: string): string[] { export interface CursorEffortTableDeps { platform: string; - stat(path: string): { mtimeMs: number; size: number } | null; - readText(path: string): string | null; + readBundle(path: string, cached?: { mtimeMs: number; size: number }): { mtimeMs: number; size: number; text: string | null } | null; } export function realCursorEffortTableDeps(): CursorEffortTableDeps { return { platform: process.platform, - stat: path => { try { const s = statSync(path); return { mtimeMs: s.mtimeMs, size: s.size }; } catch { return null; } }, - readText: path => { try { return readFileSync(path, "utf8"); } catch { return null; } }, + readBundle: readCursorBundle, }; } -let cache: { key: string; table: CursorEffortTable | null } | null = null; +function readCursorBundle(path: string, cached?: { mtimeMs: number; size: number }): { mtimeMs: number; size: number; text: string | null } | null { + let fd: number | null = null; + try { + // O_NOFOLLOW binds the validation and read to the same regular file. O_NONBLOCK + // keeps opening a substituted special file from stalling before fstat rejects it. + fd = openSync(path, constants.O_RDONLY | constants.O_NOFOLLOW | constants.O_NONBLOCK); + const stat = fstatSync(fd); + if (!stat.isFile() || stat.size > BUNDLE_MAX_BYTES) return null; + if (cached?.mtimeMs === stat.mtimeMs && cached.size === stat.size) { + return { mtimeMs: stat.mtimeMs, size: stat.size, text: null }; + } + + const chunks: Buffer[] = []; + let size = 0; + while (size <= BUNDLE_MAX_BYTES) { + const chunk = Buffer.allocUnsafe(Math.min(64 * 1024, BUNDLE_MAX_BYTES + 1 - size)); + const bytesRead = readSync(fd, chunk, 0, chunk.length, null); + if (bytesRead === 0) break; + chunks.push(chunk.subarray(0, bytesRead)); + size += bytesRead; + } + if (size > BUNDLE_MAX_BYTES) return null; + return { mtimeMs: stat.mtimeMs, size, text: Buffer.concat(chunks, size).toString("utf8") }; + } catch { + return null; + } finally { + if (fd !== null) closeSync(fd); + } +} + +let cache: { key: string; path: string; mtimeMs: number; size: number; table: CursorEffortTable | null } | null = null; /** Table from the Private Inference install, else null (caller falls back to the static mirror). */ export function loadCursorEffortTable(install: CursorInstall | undefined, deps: CursorEffortTableDeps = realCursorEffortTableDeps()): CursorEffortTable | null { if (!install) return null; const bundlePath = cursorAgentBundlePath(install, deps.platform); - const st = deps.stat(bundlePath); - if (!st || st.size > BUNDLE_MAX_BYTES) return null; - const key = `${bundlePath}|${st.mtimeMs}|${st.size}`; - if (cache?.key === key) return cache.table; - const text = deps.readText(bundlePath); - const parsed = text ? parseCursorEffortTable(text) : null; + const cachedMetadata = cache?.path === bundlePath + ? { mtimeMs: cache.mtimeMs, size: cache.size } + : undefined; + const bundle = deps.readBundle(bundlePath, cachedMetadata); + if (!bundle) return null; + const key = `${bundlePath}|${bundle.mtimeMs}|${bundle.size}`; + if (cache?.key === key) { + // The cache key covers bundle identity only; install.version comes from + // product.json and can change or resolve without touching the bundle. + const cached = cache.table; + return cached && cached.version !== install.version + ? { ...cached, version: install.version } + : cached; + } + const parsed = bundle.text ? parseCursorEffortTable(bundle.text) : null; const table = parsed ? { ...parsed, version: install.version, bundlePath } : null; - cache = { key, table }; + cache = { key, path: bundlePath, mtimeMs: bundle.mtimeMs, size: bundle.size, table }; return table; } diff --git a/structure/clients/integrations.md b/structure/clients/integrations.md index 277bc619912..92da578aa76 100644 --- a/structure/clients/integrations.md +++ b/structure/clients/integrations.md @@ -29,6 +29,14 @@ parsing and ownership rules below. | `src/integrations/mutation-plan.ts` | The shared observation both a preview and a mutation read, and the value-free plan an operator confirms. It owns no IO of its own, takes no lock, and must never import `writer.ts`. | | `src/integrations/store.ts` / `journal.ts` | One-root persistence for ownership records, operation history, snapshots, and retention maintenance. | +## Cursor installed capability reads + +`src/integrations/cursor-effort-table.ts` reads the installed agent bundle through one regular-file +handle, refuses final symlinks where supported, and caps bytes read even if the file grows after +inspection. Failure retains the static-table fallback. Parsed content is cached by path, mtime and +size; the returned table always uses the current install version, including on a cache hit. +`tests/providers/cursor/cursor-effort-table.test.ts` covers cache reuse, version refresh and unsafe files. + ## Data Flow ```text diff --git a/structure/providers/cursor.md b/structure/providers/cursor.md index 30ef6cfb9f8..28c872d566a 100644 --- a/structure/providers/cursor.md +++ b/structure/providers/cursor.md @@ -161,9 +161,10 @@ markers up to a byte-counted cap, then switches to a constant-space suppressed s the JSON object closes; neither an oversized tail nor a malformed payload returns to prose. Malformed argument diagnostics contain only the failure class and an optional tool name, never the argument content. `src/adapters/cursor/protobuf-events.ts` buffers advertised textual calls -until turn finalization. It flushes them onto the atomic tool-call path only when the turn +until turn finalization, charging each retained argument immediately against the normal per-call +and per-turn translator budgets. It flushes them onto the atomic tool-call path only when the turn contained no real client-tool frame; any real frame, including one left incomplete, wins and -drops the whole textual buffer. A missing advertised-name set is fail-closed. Finalize also +drops the whole textual buffer and releases its charges. A missing advertised-name set is fail-closed. Finalize also clears any held or suppressed prefix. Coverage lives in `tests/providers/cursor/cursor-protobuf-events.test.ts`. diff --git a/tests/providers/cursor/cursor-effort-table.test.ts b/tests/providers/cursor/cursor-effort-table.test.ts index 973941d5744..828c3bec5cf 100644 --- a/tests/providers/cursor/cursor-effort-table.test.ts +++ b/tests/providers/cursor/cursor-effort-table.test.ts @@ -1,7 +1,10 @@ import { beforeEach, describe, expect, test } from "bun:test"; -import { readFileSync } from "node:fs"; +import { execFileSync } from "node:child_process"; +import { mkdirSync, readFileSync, rmSync, symlinkSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; import { join } from "node:path"; import { + cursorAgentBundlePath, loadCursorEffortTable, parseCursorEffortTable, resetCursorEffortTableCacheForTests, @@ -65,15 +68,13 @@ describe("Cursor installed-bundle effort table", () => { test("activates the static fallback for missing installs, missing literals, and malformed regexes", () => { const missingStat: CursorEffortTableDeps = { platform: "darwin", - stat: () => null, - readText: () => { throw new Error("readText must not run without a stat"); }, + readBundle: () => null, }; expect(loadCursorEffortTable(INSTALL, missingStat)).toBeNull(); const loadSource = (source: string, mtimeMs: number) => loadCursorEffortTable(INSTALL, { platform: "darwin", - stat: () => ({ mtimeMs, size: source.length }), - readText: () => source, + readBundle: () => ({ mtimeMs, size: source.length, text: source }), }); expect(loadSource("function unrelated(){}", 1)).toBeNull(); expect(loadSource(FIXTURE.replace("/^claude-opus-5$/u", "/[/u"), 2)).toBeNull(); @@ -102,10 +103,12 @@ describe("Cursor installed-bundle effort table", () => { let reads = 0; const deps: CursorEffortTableDeps = { platform: "darwin", - stat: () => ({ mtimeMs, size: FIXTURE.length }), - readText: () => { + readBundle: (_path, cached) => { + if (cached?.mtimeMs === mtimeMs && cached.size === FIXTURE.length) { + return { ...cached, text: null }; + } reads += 1; - return FIXTURE; + return { mtimeMs, size: FIXTURE.length, text: FIXTURE }; }, }; expect(loadCursorEffortTable(INSTALL, deps)?.families).toHaveLength(16); @@ -115,4 +118,33 @@ describe("Cursor installed-bundle effort table", () => { expect(loadCursorEffortTable(INSTALL, deps)?.families).toHaveLength(16); expect(reads).toBe(2); }); + + test("refreshes the reported version on a bundle cache hit", () => { + const deps: CursorEffortTableDeps = { + platform: "darwin", + readBundle: () => ({ mtimeMs: 1, size: FIXTURE.length, text: FIXTURE }), + }; + expect(loadCursorEffortTable(INSTALL, deps)?.version).toBe("3.18.25"); + const upgraded = { ...INSTALL, version: "3.19.0" }; + expect(loadCursorEffortTable(upgraded, deps)?.version).toBe("3.19.0"); + }); + + test("rejects symlinks and special files without blocking", () => { + if (process.platform === "win32") return; + const root = `${tmpdir()}/ocx-cursor-bundle-${process.pid}-${Date.now()}`; + const install = { ...INSTALL, path: root }; + const bundlePath = cursorAgentBundlePath(install, process.platform); + const target = `${root}/target.js`; + mkdirSync(bundlePath.slice(0, bundlePath.lastIndexOf("/")), { recursive: true }); + writeFileSync(target, FIXTURE); + try { + symlinkSync(target, bundlePath); + expect(loadCursorEffortTable(install)).toBeNull(); + rmSync(bundlePath); + execFileSync("mkfifo", [bundlePath]); + expect(loadCursorEffortTable(install)).toBeNull(); + } finally { + rmSync(root, { recursive: true, force: true }); + } + }); }); diff --git a/tests/providers/cursor/cursor-protobuf-events.test.ts b/tests/providers/cursor/cursor-protobuf-events.test.ts index 4dae165c066..3c3b2e2eaf1 100644 --- a/tests/providers/cursor/cursor-protobuf-events.test.ts +++ b/tests/providers/cursor/cursor-protobuf-events.test.ts @@ -1257,7 +1257,8 @@ describe("textual pseudo tool-call marker quarantine", () => { }); test("a real frame wins over a textual echo in the same turn", () => { - const state = createCursorProtobufEventState({ clientToolNames: ["grep"] }); + const budget = createTranslatorBudget(); + const state = createCursorProtobufEventState({ clientToolNames: ["grep"], translatorBudget: budget }); expect(mapCursorProtobufServerMessage( textDelta('[TOOL_CALL]grep[ARGS]{"pattern":"echo"}'), state, @@ -1272,6 +1273,31 @@ describe("textual pseudo tool-call marker quarantine", () => { { type: "tool_call_start", id: "call_1", name: "grep" }, ]); expect(events.some(event => event.type === "tool_call_delta" && event.arguments.includes("echo"))).toBe(false); + expect(budget.snapshot().currentBytes).toBe(0); + budget.dispose(); + }); + + test("complete textual fallbacks are budgeted before they are retained", () => { + const budget = createTranslatorBudget({ maxCallArgumentBytes: 32, maxTurnBytes: 40 }); + const state = createCursorProtobufEventState({ clientToolNames: ["grep"], translatorBudget: budget }); + try { + expect(() => mapCursorProtobufServerMessage( + textDelta(`[TOOL_CALL]grep[ARGS]{"pattern":"${"x".repeat(40)}"}`), + state, + )).toThrow("translator tool_args buffer exceeded 32 bytes"); + expect(state.bufferedTextToolCalls).toBeUndefined(); + expect(budget.snapshot().currentBytes).toBe(0); + + mapCursorProtobufServerMessage(textDelta('[TOOL_CALL]grep[ARGS]{"pattern":"12345678"}'), state); + expect(budget.snapshot().currentBytes).toBeGreaterThan(0); + expect(() => mapCursorProtobufServerMessage( + textDelta('[TOOL_CALL]grep[ARGS]{"pattern":"abcdefgh"}'), + state, + )).toThrow("translator tool_args buffer exceeded 40 bytes"); + expect(state.bufferedTextToolCalls).toHaveLength(1); + } finally { + budget.dispose(); + } }); test("a split marker is dropped when an incomplete real frame appears", () => { From 59f9913b2a37c95d95351fb4b0de4fcad6f33332 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:04:01 +0900 Subject: [PATCH 02/24] fix(moonshot): bound normalized tool-schema expansion (#5547) Carries #5547, which consolidates #5464 and the request-wide inline budget, onto current dev. Co-authored-by: yeongjunyoo <47925973+yeongjunyoo@users.noreply.github.com> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Co-authored-by: Epinephrine --- src/adapters/openai-chat/tool-schema.ts | 117 +++++++++++- ...55-chat-structured-output-compatibility.md | 27 +++ structure/providers/chat-compat.md | 9 +- tests/providers/moonshot-tool-schema.test.ts | 178 +++++++++++++++++- 4 files changed, 324 insertions(+), 7 deletions(-) create mode 100644 structure/decisions/ADR-0355-chat-structured-output-compatibility.md diff --git a/src/adapters/openai-chat/tool-schema.ts b/src/adapters/openai-chat/tool-schema.ts index 23f77121bd5..4e3d1834a07 100644 --- a/src/adapters/openai-chat/tool-schema.ts +++ b/src/adapters/openai-chat/tool-schema.ts @@ -198,6 +198,37 @@ const MOONSHOT_MAX_REF_EXPANSIONS = 512; */ const MOONSHOT_MAX_SCHEMA_DEPTH = 64; const MOONSHOT_MAX_SCHEMA_NODES = 4_096; +const MOONSHOT_MAX_INLINED_SCHEMA_BYTES = 1024 * 1024; + +/** + * Measure only as far as the caller's remaining allowance. Keeping this iterative avoids + * reintroducing the deep-schema stack exhaustion that the normalizer's depth limit prevents. + */ +function serializedJsonBytesUpTo(value: unknown, limit: number): number { + const encoder = new TextEncoder(); + const pending: unknown[] = [value]; + let bytes = 0; + while (pending.length > 0 && bytes <= limit) { + const item = pending.pop(); + if (Array.isArray(item)) { + bytes += 2 + Math.max(0, item.length - 1); + for (const child of item) pending.push(child); + continue; + } + if (isXaiObjectSchema(item)) { + const entries = Object.entries(item); + bytes += 2 + Math.max(0, entries.length - 1); + for (const [key, child] of entries) { + bytes += encoder.encode(JSON.stringify(key)).byteLength + 1; + pending.push(child); + } + continue; + } + const encoded = JSON.stringify(item); + bytes += encoder.encode(encoded === undefined ? "null" : encoded).byteLength; + } + return bytes; +} /** * Assertion keywords whose meaning under a `$ref` is CONJUNCTION, not replacement. A node @@ -309,8 +340,19 @@ function composeProperties( return combined; } +/** + * The inline-byte allowance for one request. Sharing it across tools matters: a per-tool + * budget would let a large catalog multiply the cap by its tool count, reintroducing the + * request amplification this bound exists to prevent. + */ +interface MoonshotInlineByteBudget { + remaining: number; +} + interface MoonshotNormalizeState { activeRefs: Set; + inlineSizeCache: WeakMap, number>; + inlineByteBudget: MoonshotInlineByteBudget; remainingExpansions: number; remainingNodes: number; } @@ -342,10 +384,28 @@ function normalizeMoonshotSchemaNode( const target = lookupLocalJsonPointer(root, ref); if (isXaiObjectSchema(target)) { + // Charge the referenced value before copying it. Object/node counts do not cover large + // maps of boolean schemas, which otherwise allow a small input to create hundreds of + // full copies before the final request is serialized. + let inlineBytes = state.inlineSizeCache.get(target); + if (inlineBytes === undefined) { + inlineBytes = serializedJsonBytesUpTo(target, MOONSHOT_MAX_INLINED_SCHEMA_BYTES); + state.inlineSizeCache.set(target, inlineBytes); + } + if (inlineBytes > state.inlineByteBudget.remaining) return { $ref: ref }; + state.inlineByteBudget.remaining -= inlineBytes; state.remainingExpansions -= 1; state.activeRefs.add(ref); const resolvedTarget = normalizeMoonshotSchemaNode(target, root, state, depth + 1); state.activeRefs.delete(ref); + // Type inference and nested normalization can enlarge the raw target we reserved. + // Charge that growth before retaining the copy; nested expansions share this allowance. + const normalizedBytes = serializedJsonBytesUpTo( + resolvedTarget, inlineBytes + state.inlineByteBudget.remaining, + ); + const growthBytes = Math.max(0, normalizedBytes - inlineBytes); + if (growthBytes > state.inlineByteBudget.remaining) return { $ref: ref }; + state.inlineByteBudget.remaining -= growthBytes; const merged: Record = Object.create(null) as Record; if (isXaiObjectSchema(resolvedTarget)) { for (const [key, value] of Object.entries(resolvedTarget)) merged[key] = value; @@ -380,6 +440,20 @@ function normalizeMoonshotSchemaNode( } merged[key] = normalized; } + + // Re-normalize only composed properties that retain a $ref alongside sibling keywords + if (isXaiObjectSchema(merged.properties)) { + for (const [propName, propVal] of Object.entries(merged.properties as Record)) { + if (isXaiObjectSchema(propVal) && typeof propVal.$ref === "string" && moonshotRefTargetKeys(propVal).length > 0) { + (merged.properties as Record)[propName] = normalizeMoonshotSchemaNode( + propVal, + root, + state, + depth + 1, + ); + } + } + } return merged; } @@ -397,13 +471,49 @@ function normalizeMoonshotSchemaNode( ? value : normalizeMoonshotSchemaNode(value, root, state, depth + 1); } + + // Moonshot MFJS requirements: + // 1. Stamp "object" if properties are present, or if allOf defines object properties/variants, + // so Moonshot's validator recognizes the schema as a valid termination condition. + // 2. Infer scalar types for bare const and enum keywords. + if (out.type === undefined) { + const isObjectAllOf = Array.isArray(out.allOf) && out.allOf.some( + variant => isXaiObjectSchema(variant) && ( + variant.type === "object" || + variant.properties !== undefined || + variant.additionalProperties !== undefined + ), + ); + if (out.properties !== undefined || out.additionalProperties !== undefined || isObjectAllOf) { + out.type = "object"; + } else if (out.const !== undefined) { + const t = typeof out.const; + if (t === "string" || t === "number" || t === "boolean") { + out.type = t; + } + } else if (Array.isArray(out.enum) && out.enum.length > 0) { + if (out.enum.every(x => typeof x === "string")) { + out.type = "string"; + } else if (out.enum.every(x => typeof x === "number")) { + out.type = "number"; + } else if (out.enum.every(x => typeof x === "boolean")) { + out.type = "boolean"; + } + } + } + return out; } -function normalizeMoonshotToolParameters(parameters: unknown): Record { +function normalizeMoonshotToolParameters( + parameters: unknown, + inlineByteBudget: MoonshotInlineByteBudget, +): Record { const rooted = ensureRootObjectType(parameters); const normalized = normalizeMoonshotSchemaNode(rooted, rooted, { activeRefs: new Set(), + inlineSizeCache: new WeakMap, number>(), + inlineByteBudget, remainingExpansions: MOONSHOT_MAX_REF_EXPANSIONS, remainingNodes: MOONSHOT_MAX_SCHEMA_NODES, }); @@ -420,11 +530,14 @@ export function toolsToChatFormat( if (tools.length === 0) return undefined; const xaiTarget = isXaiSchemaTarget(provider); const moonshotTarget = !xaiTarget && isMoonshotSchemaTarget(provider); + const moonshotInlineByteBudget: MoonshotInlineByteBudget = { + remaining: MOONSHOT_MAX_INLINED_SCHEMA_BYTES, + }; const formatted = tools.flatMap(t => { const normalized = xaiTarget ? normalizeXaiToolParameters(t.parameters) : moonshotTarget - ? normalizeMoonshotToolParameters(t.parameters) + ? normalizeMoonshotToolParameters(t.parameters, moonshotInlineByteBudget) : ensureRootObjectType(t.parameters); const parameters = stripUnicodePropertyPatterns(stripResponsesOnlyEncryptedMarker(normalized)); diff --git a/structure/decisions/ADR-0355-chat-structured-output-compatibility.md b/structure/decisions/ADR-0355-chat-structured-output-compatibility.md new file mode 100644 index 00000000000..661d7457c45 --- /dev/null +++ b/structure/decisions/ADR-0355-chat-structured-output-compatibility.md @@ -0,0 +1,27 @@ +# ADR-0355 — decision recorded under "Chat structured-output compatibility" + +- Contract owner: [providers/chat-compat.md](../providers/chat-compat.md#chat-structured-output-compatibility) + +## Decision record + +- 목적과 의도: Bound the request amplification a Moonshot `$ref` inlining can produce without + weakening the tool schema beyond what the wire forces. +- 기존 구현 및 제약 조건: The normalizer walks depth-, node-, and expansion-bounded, but a small + input can name a large boolean `properties` map from many nodes, so each bound can pass while + the serialized output still repeats the map hundreds of times. The adapter sits on the request + path, so amplification is user-facing latency and payload size. +- 검토한 주요 대안: (1) Keep only the three existing budgets. (2) Measure the final serialized + request and reject it over a size cap. (3) Charge each inlined target its serialized JSON bytes + against a shared byte budget before copying it. +- 선택한 방식: (3). Each expansion measures the referenced schema's serialized size once per + target object, charges it against one 1 MiB allowance shared by every tool in the request, + and a reference that would exceed the remaining allowance stays a bare `$ref`. +- 다른 대안 대신 이 방식을 선택한 이유: (1) leaves the demonstrated amplification reachable — + node and expansion counts stay small while output grows without bound. (2) detects the blow-up + only after the bytes were already produced, and a whole-request rejection discards a schema + Moonshot would have accepted in partially inlined form. +- 장점, 단점 및 영향: Output size is bounded independently of how the reference graph is shaped, + and over-budget nodes degrade to the same bare-`$ref` fallback the other budgets already use. + Measuring is iterative and capped at the remaining allowance, so the guard itself cannot + reintroduce the deep-schema stack exhaustion the depth budget prevents. Moonshot 계열 + `openai-chat` baseUrl에만 적용되고 다른 provider는 손대지 않는다. diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index 1c7465bde6e..7ec0ef7ba66 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -308,10 +308,15 @@ First-party Kimi and Moonshot Chat destinations normalize a `$ref` with sibling their wire rejects that valid JSON Schema 2020-12 shape. Inlining preserves conjunction semantics: `required` members are unioned, lower numeric bounds take the maximum, upper numeric bounds take the minimum, and overlapping `properties` recurse with the same rules. The walk remains depth-, node-, -and expansion-bounded. Unresolvable or cyclic references keep the existing bare-`$ref` fallback, -and unrelated OpenAI-compatible providers retain the caller's schema unchanged. +expansion-, and inline-byte-bounded: each inlined reference is charged its serialized size against +one 1 MiB allowance shared by every tool in the request. Raw target bytes are reserved before +normalization; inferred types and nested normalization must also fit the remaining allowance +before their copy is retained. An over-budget reference keeps the existing bare-`$ref` fallback. +Unresolvable or cyclic references do the same, and unrelated OpenAI-compatible providers +retain the caller's schema unchanged. > Decision record: [ADR-0064](../decisions/ADR-0064-chat-structured-output-compatibility.md) +> Decision record: [ADR-0355](../decisions/ADR-0355-chat-structured-output-compatibility.md) The `openai-chat` adapter translates Responses `text.format` and Chat Completions `response_format` through one internal format, then emits `response_format` on the upstream chat diff --git a/tests/providers/moonshot-tool-schema.test.ts b/tests/providers/moonshot-tool-schema.test.ts index ed1626927ad..f8b257484da 100644 --- a/tests/providers/moonshot-tool-schema.test.ts +++ b/tests/providers/moonshot-tool-schema.test.ts @@ -7,12 +7,12 @@ const createOpenAIChatAdapter = ( ...args: Parameters ) => withTestTranslatorBudget(createOpenAIChatAdapterProduction(...args)); -function parsedRequest(tool: OcxTool): OcxParsedRequest { +function parsedRequest(tool: OcxTool | OcxTool[]): OcxParsedRequest { return { modelId: "k3", context: { messages: [{ role: "user", content: "run the tool", timestamp: 0 }], - tools: [tool], + tools: [tool].flat(), }, stream: true, options: {}, @@ -185,7 +185,6 @@ describe("Moonshot tool schema normalization (issue #2673)", () => { expect(properties.value).toEqual({ $ref: "https://example.com/schema.json#/Thing" }); }); - test("composes duplicate required, properties, and same-key assertions", async () => { // The reviewer's first blocker. `$ref` under 2020-12 is an in-place applicator: the // node and its target BOTH apply. Overwriting made a tool that required `a` and `b` @@ -305,6 +304,77 @@ describe("Moonshot tool schema normalization (issue #2673)", () => { expect(siblingRefPaths(parameters)).toEqual([]); }); + test("bounds repeated large property-map inlining by serialized bytes", async () => { + const bigProperties = Object.fromEntries( + Array.from({ length: 10_000 }, (_, index) => [`property_${index}`, true]), + ); + const references = Object.fromEntries( + Array.from({ length: 64 }, (_, index) => [ + `value_${index}`, + { $ref: "#/$defs/Big", properties: { sibling: { type: "string" } } }, + ]), + ); + const tool: OcxTool = { + name: "bounded_amplification_tool", + parameters: { + type: "object", + $defs: { Big: { type: "object", properties: bigProperties } }, + properties: references, + }, + }; + + const request = await adapterFor("https://api.moonshot.ai/v1").buildRequest(parsedRequest(tool)); + const inputBytes = new TextEncoder().encode(JSON.stringify(tool.parameters)).byteLength; + const outputBytes = new TextEncoder().encode(request.body).byteLength; + const parameters = JSON.parse(request.body).tools[0].function.parameters as Record; + + // The original definition remains available, but repeated sibling refs stop inlining once + // their cumulative serialized cost reaches the fixed allowance. + expect(outputBytes).toBeLessThan(inputBytes + 2 * 1024 * 1024); + expect(siblingRefPaths(parameters)).toEqual([]); + const emitted = parameters.properties as Record>; + expect(Object.values(emitted).some(value => Object.keys(value).length === 1 && "$ref" in value)).toBe(true); + }); + + test("shares the inline-byte budget across the tools of one request", async () => { + // A per-tool allowance would multiply the cap by the catalog size: the second tool + // must spend what the first already charged. + const bigTool = (name: string): OcxTool => ({ + name, + parameters: { + type: "object", + $defs: { + Big: { + type: "object", + properties: Object.fromEntries( + Array.from({ length: 30_000 }, (_, index) => [`property_${index}`, true]), + ), + }, + }, + properties: { + a: { $ref: "#/$defs/Big", properties: { s: { type: "string" } } }, + b: { $ref: "#/$defs/Big", properties: { s: { type: "string" } } }, + }, + }, + }); + + const request = await adapterFor("https://api.moonshot.ai/v1").buildRequest( + parsedRequest([bigTool("first_tool"), bigTool("second_tool")]), + ); + const tools = (JSON.parse(request.body) as { + tools: { function: { parameters: { properties: Record> } } }[]; + }).tools; + const bareRefCount = (tool: (typeof tools)[number]) => + Object.values(tool.function.parameters.properties).filter( + value => Object.keys(value).length === 1 && "$ref" in value, + ).length; + + // Each inline costs ~0.6 MB of the shared 1 MiB allowance, so only the first of the + // four sibling refs fits; the rest degrade to the bare-$ref fallback. + expect(bareRefCount(tools[0])).toBe(1); + expect(bareRefCount(tools[1])).toBe(2); + }); + test("composes a property that both the target and the node define", async () => { // The same conjunction problem `required` had, one level down. Letting the sibling @@ -325,6 +395,42 @@ describe("Moonshot tool schema normalization (issue #2673)", () => { expect(shared.type).toBe("string"); }); + test("counts type-inference growth before retaining an inlined target", async () => { + const target = { + properties: Object.fromEntries(Array.from({ length: 1_000 }, (_, index) => [`p${index}`, { const: "v" }])), + description: "", + }; + target.description = "x".repeat(1024 * 1024 - JSON.stringify(target).length - 100); + const parameters = await emittedParameters("https://api.kimi.com/coding/v1", { + name: "inferred_byte_growth", + parameters: { type: "object", $defs: { Big: target }, properties: { value: { $ref: "#/$defs/Big", required: ["p0"] } } }, + }); + const value = (parameters!.properties as Record>).value; + // Raw target bytes fit, but the inferred types push the copied target over 1 MiB. + expect(Object.keys(value)).toEqual(["$ref"]); + expect(value.$ref).toBe("#/$defs/Big"); + const definition = (parameters!.$defs as Record).Big; + expect(definition.properties.p0).toMatchObject({ const: "v", type: "string" }); + }); + + test("composed-property re-normalization spends the shared catalog byte allowance", async () => { + const bigProperties = Object.fromEntries(Array.from({ length: 30_000 }, (_, index) => [`property_${index}`, true])); + const tool = (name: string): OcxTool => ({ name, parameters: { + type: "object", + $defs: { Big: { type: "object", properties: bigProperties }, Base: { type: "object", properties: { child: { $ref: "#/$defs/Big" } } } }, + properties: { value: { $ref: "#/$defs/Base", properties: { child: { properties: { sibling: { type: "string" } } } } } }, + } }); + const request = await adapterFor("https://api.kimi.com/coding/v1").buildRequest(parsedRequest([tool("first"), tool("second")])); + const emitted = JSON.parse(request.body).tools; + const first = emitted[0].function.parameters.properties.value.properties.child; + const second = emitted[1].function.parameters.properties.value.properties.child; + expect(first.properties.sibling).toEqual({ type: "string" }); + expect(first.properties.property_0).toBe(true); + expect(Object.keys(second)).toEqual(["$ref"]); + expect(second.$ref).toBe("#/$defs/Big"); + expect(siblingRefPaths(emitted)).toEqual([]); + }); + test("intersects bounds when both sides define the same property", async () => { const parameters = await emittedParameters("https://api.moonshot.ai/v1", { name: "shared_property_bounds_tool", @@ -447,4 +553,70 @@ describe("Moonshot tool schema normalization (issue #2673)", () => { expect(parameters?.$defs).toEqual(CODEX_STYLE_SCHEMA.$defs as Record); expect(siblingRefPaths(parameters).length).toBeGreaterThan(0); }); + + test("infers object type for allOf/properties and scalar types for const/enum", async () => { + const parameters = await emittedParameters("https://api.kimi.com/coding/v1", { + name: "inference_tool", + parameters: { + type: "object", + properties: { + leaf: { + allOf: [{ properties: { id: { type: "integer" } } }], + }, + status: { const: "ACTIVE" }, + count: { const: 42 }, + flag: { const: true }, + color: { enum: ["red", "blue"] }, + toggle: { enum: [true, false] }, + stringAllOf: { allOf: [{ type: "string" }, { minLength: 1 }] }, + }, + }, + }); + + const props = parameters?.properties as Record>; + expect(props.leaf.type).toBe("object"); + expect(props.status.type).toBe("string"); + expect(props.count.type).toBe("number"); + expect(props.flag.type).toBe("boolean"); + expect(props.color.type).toBe("string"); + expect(props.toggle.type).toBe("boolean"); + expect(props.stringAllOf.type).toBeUndefined(); + }); + + test("re-normalizes composed properties when sibling narrows a referenced property", async () => { + // When Base defines `op: { $ref: "#/$defs/Op" }` and a sibling node narrows it with + // `properties: { op: { const: "AND" } }`, `composeProperties` merges them. The merged + // property must re-normalize rather than emitting a `$ref` beside `const`. + const parameters = await emittedParameters("https://api.kimi.com/coding/v1", { + name: "ast_tool", + parameters: { + type: "object", + $defs: { + Op: { type: "string", enum: ["AND", "OR"] }, + Base: { + type: "object", + properties: { + op: { $ref: "#/$defs/Op" }, + left: { type: "string" }, + }, + }, + }, + properties: { + andNode: { + $ref: "#/$defs/Base", + properties: { + op: { const: "AND" }, + }, + }, + }, + }, + }); + + expect(siblingRefPaths(parameters)).toEqual([]); + const andNode = (parameters?.properties as Record>).andNode; + const op = (andNode.properties as Record>).op; + expect(op.$ref).toBeUndefined(); + expect(op.const).toBe("AND"); + expect(op.type).toBe("string"); + }); }); From 651d53bb4346a10585f5b5fc78250b4bfb0ecd4f Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 06:10:30 +0900 Subject: [PATCH 03/24] fix(moonshot): restore rejected inline budgets and charge nested growth once A rejected sibling-reference expansion now restores the byte, node and expansion allowances it consumed, and outer growth no longer re-charges nested copies, so later independent expansions in the same request keep their allowance. Documents the provider-driven object type inference as a deliberate tradeoff and rewrites ADR-0355 in English. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- src/adapters/openai-chat/tool-schema.ts | 19 ++++-- ...55-chat-structured-output-compatibility.md | 38 +++++------ structure/providers/chat-compat.md | 10 ++- tests/providers/moonshot-tool-schema.test.ts | 65 +++++++++++++++++++ 4 files changed, 104 insertions(+), 28 deletions(-) diff --git a/src/adapters/openai-chat/tool-schema.ts b/src/adapters/openai-chat/tool-schema.ts index 4e3d1834a07..641014a0066 100644 --- a/src/adapters/openai-chat/tool-schema.ts +++ b/src/adapters/openai-chat/tool-schema.ts @@ -393,18 +393,27 @@ function normalizeMoonshotSchemaNode( state.inlineSizeCache.set(target, inlineBytes); } if (inlineBytes > state.inlineByteBudget.remaining) return { $ref: ref }; + const bytesBefore = state.inlineByteBudget.remaining; + const expansionsBefore = state.remainingExpansions; + const nodesBefore = state.remainingNodes; state.inlineByteBudget.remaining -= inlineBytes; state.remainingExpansions -= 1; state.activeRefs.add(ref); const resolvedTarget = normalizeMoonshotSchemaNode(target, root, state, depth + 1); state.activeRefs.delete(ref); - // Type inference and nested normalization can enlarge the raw target we reserved. - // Charge that growth before retaining the copy; nested expansions share this allowance. + // Nested copies have already spent from the shared allowance. Charge only + // growth that their own charges do not cover. + const nestedCharges = bytesBefore - inlineBytes - state.inlineByteBudget.remaining; const normalizedBytes = serializedJsonBytesUpTo( - resolvedTarget, inlineBytes + state.inlineByteBudget.remaining, + resolvedTarget, bytesBefore, ); - const growthBytes = Math.max(0, normalizedBytes - inlineBytes); - if (growthBytes > state.inlineByteBudget.remaining) return { $ref: ref }; + const growthBytes = Math.max(0, normalizedBytes - inlineBytes - nestedCharges); + if (growthBytes > state.inlineByteBudget.remaining) { + state.inlineByteBudget.remaining = bytesBefore; + state.remainingExpansions = expansionsBefore; + state.remainingNodes = nodesBefore; + return { $ref: ref }; + } state.inlineByteBudget.remaining -= growthBytes; const merged: Record = Object.create(null) as Record; if (isXaiObjectSchema(resolvedTarget)) { diff --git a/structure/decisions/ADR-0355-chat-structured-output-compatibility.md b/structure/decisions/ADR-0355-chat-structured-output-compatibility.md index 661d7457c45..1c22c361449 100644 --- a/structure/decisions/ADR-0355-chat-structured-output-compatibility.md +++ b/structure/decisions/ADR-0355-chat-structured-output-compatibility.md @@ -4,24 +4,20 @@ ## Decision record -- 목적과 의도: Bound the request amplification a Moonshot `$ref` inlining can produce without - weakening the tool schema beyond what the wire forces. -- 기존 구현 및 제약 조건: The normalizer walks depth-, node-, and expansion-bounded, but a small - input can name a large boolean `properties` map from many nodes, so each bound can pass while - the serialized output still repeats the map hundreds of times. The adapter sits on the request - path, so amplification is user-facing latency and payload size. -- 검토한 주요 대안: (1) Keep only the three existing budgets. (2) Measure the final serialized - request and reject it over a size cap. (3) Charge each inlined target its serialized JSON bytes - against a shared byte budget before copying it. -- 선택한 방식: (3). Each expansion measures the referenced schema's serialized size once per - target object, charges it against one 1 MiB allowance shared by every tool in the request, - and a reference that would exceed the remaining allowance stays a bare `$ref`. -- 다른 대안 대신 이 방식을 선택한 이유: (1) leaves the demonstrated amplification reachable — - node and expansion counts stay small while output grows without bound. (2) detects the blow-up - only after the bytes were already produced, and a whole-request rejection discards a schema - Moonshot would have accepted in partially inlined form. -- 장점, 단점 및 영향: Output size is bounded independently of how the reference graph is shaped, - and over-budget nodes degrade to the same bare-`$ref` fallback the other budgets already use. - Measuring is iterative and capped at the remaining allowance, so the guard itself cannot - reintroduce the deep-schema stack exhaustion the depth budget prevents. Moonshot 계열 - `openai-chat` baseUrl에만 적용되고 다른 provider는 손대지 않는다. +- Purpose: Bound Moonshot reference-inlining amplification while retaining the tool schema + whenever an individual expansion fits. +- Prior constraint: Depth, node, and expansion counts did not bound repeated copies of a + large referenced value across one tool catalog. The normalizer runs on the request path. +- Alternatives: Keep the three existing bounds; reject the whole serialized request when + large; or reserve each candidate's copied bytes against one request-shared allowance. +- Choice: Reserve raw target bytes before normalization. Charge nested retained copies once + and then only additional outer growth. Restore byte, node, and expansion allowances when + a candidate falls back to a bare reference. Share a 1 MiB allowance across the request. +- Reason: Whole-request rejection happens after allocation and discards independent valid + tool schemas. A transactional candidate keeps later expansions available. +- Consequences: The bound applies only to Moonshot-family Chat targets. Its validator + requires an explicit object termination type for recursive unions, so a schema carrying + properties or additionalProperties, or an allOf with such a member, is emitted with + `type: "object"`. This narrows scalar instances JSON Schema would permit; tool-argument + schemas do not rely on them. Non-object allOf compositions remain untyped. Unrelated + providers keep their schemas. diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index 7ec0ef7ba66..a14c7c24d98 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -310,8 +310,14 @@ their wire rejects that valid JSON Schema 2020-12 shape. Inlining preserves conj minimum, and overlapping `properties` recurse with the same rules. The walk remains depth-, node-, expansion-, and inline-byte-bounded: each inlined reference is charged its serialized size against one 1 MiB allowance shared by every tool in the request. Raw target bytes are reserved before -normalization; inferred types and nested normalization must also fit the remaining allowance -before their copy is retained. An over-budget reference keeps the existing bare-`$ref` fallback. +normalization. A candidate expansion restores its byte, node, and expansion allowances when +it falls back; nested retained copies spend the allowance once, and only additional outer growth +is charged. Moonshot's validator requires an explicit object termination type for recursive +unions, so a schema carrying `properties` or `additionalProperties`, or an `allOf` with such a +member, is emitted with `type: "object"`. This narrows scalar instances that JSON Schema would +permit; tool-argument schemas do not rely on those scalar instances. Non-object `allOf` +compositions, such as string constraints, remain untyped. An over-budget reference keeps the +bare-`$ref` fallback. Unresolvable or cyclic references do the same, and unrelated OpenAI-compatible providers retain the caller's schema unchanged. diff --git a/tests/providers/moonshot-tool-schema.test.ts b/tests/providers/moonshot-tool-schema.test.ts index f8b257484da..43bcab3728a 100644 --- a/tests/providers/moonshot-tool-schema.test.ts +++ b/tests/providers/moonshot-tool-schema.test.ts @@ -413,6 +413,71 @@ describe("Moonshot tool schema normalization (issue #2673)", () => { expect(definition.properties.p0).toMatchObject({ const: "v", type: "string" }); }); + test("restores a rejected outer candidate before later sibling and tool expansions", async () => { + const outer = { + properties: { + child: { $ref: "#/$defs/Inner", minLength: 1 }, + ...Object.fromEntries(Array.from({ length: 1_000 }, (_, index) => [`p${index}`, { const: "v" }])), + }, + description: "", + }; + outer.description = "x".repeat(1024 * 1024 - JSON.stringify(outer).length - 96); + const small = { type: "string", description: "small".repeat(24) }; + const first: OcxTool = { + name: "first", + parameters: { + type: "object", + properties: { + rejected: { $ref: "#/$defs/Outer", required: ["child"] }, + later: { $ref: "#/$defs/Small", minLength: 1 }, + }, + $defs: { Outer: outer, Inner: { type: "string", minLength: 2 }, Small: small }, + }, + }; + const second: OcxTool = { + name: "second", + parameters: { + type: "object", + properties: { later: { $ref: "#/$defs/Small", minLength: 1 } }, + $defs: { Small: small }, + }, + }; + const request = await adapterFor(MOONSHOT_HOSTS[0]!).buildRequest(parsedRequest([first, second])); + const tools = (JSON.parse(request.body) as { + tools: { function: { parameters: { properties: Record> } } }[]; + }).tools; + const firstProps = tools[0]!.function.parameters.properties; + const secondProps = tools[1]!.function.parameters.properties; + expect(firstProps.rejected).toEqual({ $ref: "#/$defs/Outer" }); + expect(firstProps.later?.type).toBe("string"); + expect(firstProps.later?.description).toBe(small.description); + expect(secondProps.later?.description).toBe(small.description); + expect(siblingRefPaths(tools)).toEqual([]); + }); + + test("does not charge nested inline bytes again as outer growth", async () => { + const parameters = await emittedParameters(MOONSHOT_HOSTS[0]!, { + name: "nested_growth", + parameters: { + type: "object", + properties: { value: { $ref: "#/$defs/Outer", required: ["child"] } }, + $defs: { + Outer: { + type: "object", + description: "o".repeat(500_000), + properties: { child: { $ref: "#/$defs/Inner", minLength: 1 } }, + }, + Inner: { type: "string", description: "i".repeat(300_000) }, + }, + }, + }); + const value = (parameters!.properties as Record>).value!; + const child = (value.properties as Record>).child!; + expect(value.type).toBe("object"); + expect(child.type).toBe("string"); + expect(child.description).toBe("i".repeat(300_000)); + }); + test("composed-property re-normalization spends the shared catalog byte allowance", async () => { const bigProperties = Object.fromEntries(Array.from({ length: 30_000 }, (_, index) => [`property_${index}`, true])); const tool = (name: string): OcxTool => ({ name, parameters: { From 267606d90c9e1f1578ee58b7bd407281a54d8ac5 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:23:44 +0900 Subject: [PATCH 04/24] feat(reasoning): consolidate replay, opt-in tag parsing, and summary policy (#5566) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Carries #5566, which consolidates #5449, #5205 and #5491, onto current dev. The provider guide keeps the current bridge replay paragraph and adds the inline-tag and summary paragraphs. Co-authored-by: Joonsuh Park Co-authored-by: Daniel Sjöstrand <16033062+Danielsjostrand1979@users.noreply.github.com> Co-authored-by: alexph-dev Co-authored-by: Yum-wu <1172989563@qq.com> --- .../docs/reference/configuration/providers.md | 9 + scripts/test-layout/layout.json | 3 + src/adapters/inline-think-tags.ts | 240 ++++++++++++++++++ src/adapters/kiro-thinking.ts | 112 -------- src/adapters/kiro/stream.ts | 4 +- src/adapters/openai-chat.ts | 16 +- src/adapters/openai-chat/messages.ts | 10 +- src/combos/request.ts | 8 +- src/providers/derive.ts | 4 + src/providers/model-rename-fields.ts | 1 + src/providers/registry/model-ids.ts | 1 + src/providers/registry/types.ts | 4 +- src/providers/resolved-model-policy.ts | 4 +- src/responses/parser.ts | 3 +- src/router.ts | 2 + src/server/auth-cors.ts | 1 + src/types/provider.ts | 10 + structure/gui-and-management-api.md | 3 + structure/providers-and-adapters.md | 3 + structure/providers/chat-compat.md | 35 ++- structure/transports/responses.md | 4 + .../openai/inline-think-boundaries.test.ts | 67 +++++ .../openai-chat-inline-think-tags.test.ts | 150 +++++++++++ tests/codex-integration/combos.test.ts | 10 +- tests/fixtures/test-layout-expected.json | 3 + .../deepseek-reasoning-replay-gaps.test.ts | 42 +++ tests/providers/kiro/kiro-stream.test.ts | 4 +- .../providers/model-rename-migration.test.ts | 3 + tests/providers/resolved-model-policy.test.ts | 8 + .../reasoning-effort-summary-default.test.ts | 85 +++++++ 30 files changed, 714 insertions(+), 135 deletions(-) create mode 100644 src/adapters/inline-think-tags.ts delete mode 100644 src/adapters/kiro-thinking.ts create mode 100644 tests/adapters/openai/inline-think-boundaries.test.ts create mode 100644 tests/adapters/openai/openai-chat-inline-think-tags.test.ts create mode 100644 tests/responses/reasoning-effort-summary-default.test.ts diff --git a/docs-site/src/content/docs/reference/configuration/providers.md b/docs-site/src/content/docs/reference/configuration/providers.md index 684ad40a6e8..571f42399da 100644 --- a/docs-site/src/content/docs/reference/configuration/providers.md +++ b/docs-site/src/content/docs/reference/configuration/providers.md @@ -236,6 +236,7 @@ Providers can expose a built-in shorthand, such as `agy` for `google-antigravity | `retryOnReset?` | `{ enabled?: boolean; replacements?: number }` | Native `openai-responses` providers, including `authMode: "forward"`. Opt-in replacement of a send that failed while the caller had observed nothing: absent means off, object presence enables it unless `enabled: false`. Covers both ambiguous stages — a connection that died before any response header, and an SSE body that died after the header while carrying only control events. Only a self-contained request is ever replaced: `store: false`, complete `input`, no `previous_response_id`, `conversation` or `stream_id`, and only client-executed tools. `replacements` is the number of replacement sends ONE logical request may make across every leg and every combo child (1..2, default 1) — not a per-leg retry count and not a send budget, so a replacement still has to fit inside the send allowance the leg already had. A request that already emitted output or a tool call is never replaced, whatever this is set to. The replacement inference may still be billed if the origin had already started the first one, which is why this is off by default. | | `autoToolChoiceOnlyModels?` | `string[]` | Models whose `tool_choice` accepts only `auto` or `none`; forced choices are downgraded. | | `preserveReasoningContentModels?` | `string[]` | Models requiring prior assistant `reasoning_content` in chat history. | +| `inlineThinkTagModels?` | `string[]` | Opt-in recovery for `openai-chat` gateways without a server-side reasoning parser. A leading `` / `` / `` block (optionally after whitespace) activates splitting in streamed and buffered replies. All answer whitespace is preserved. Subsequent tags are delimiters anywhere, including same-line interleaving and code fences; this mode does not interpret Markdown. Ordinary text or a code fence before the first tag keeps the whole reply untouched. Off by default; prefer structured upstream reasoning or `reasoningSplitModels` where supported. | | `reasoningDetailsModels?` | `string[]` | Models whose endpoint returns thinking as a structured `reasoning_details` array (MiniMax M-series with `reasoning_split`); stream deltas are cumulative snapshots that are prefix-diffed, and preserved reasoning replays as a `reasoning_details` array instead of a `reasoning_content` string. | | `requiresReasoningPlaceholderModels?` | `string[]` | Models whose upstream rejects a tool_call continuation missing `reasoning_content` (DeepSeek thinking mode); a minimal placeholder is injected when the replay cache misses. Defaults to `preserveReasoningContentModels`; set `[]` to opt out. | | `showThinkingSummary?` | `boolean` | Display provider-authored summaries when a Responses client omits `reasoning.summary`. Explicit wire `"none"` wins; a client that serializes its preference as omission cannot be distinguished. Raw reasoning remains content and is never relabeled as a summary. The `google-antigravity` preset defaults to `true`; explicit `false` disables that default. CCA Gemini requests also opt into `generationConfig.thinkingConfig.includeThoughts` when display is enabled; image, Claude and gpt-oss requests do not. This does not change client configuration or global catalog summary defaults. | @@ -268,6 +269,14 @@ the same caller, conversation, provider, model and selected key. The caller is i opencodex API key it presents, so a client that sends no opencodex API key gets no such replay: its earlier search cells reach the provider unchanged, as they do for a provider without the bridge. +An explicit `inlineThinkTagModels` list replaces matching registry defaults; `[]` disables recovery. + +Translated Responses requests with a validated active reasoning effort preserve raw reasoning +when `reasoning.summary` is omitted. Explicit `"none"` keeps it hidden for replay; omission +without an active effort also stays hidden. An injected combo default adds `summary: "auto"` +only when the caller has not chosen a summary mode. Raw reasoning is never relabeled as a +provider-authored summary, and Codex still controls its display with `show_raw_agent_reasoning`. + Custom-model `reasoningEfforts` normally override discovered provider metadata. The bounded exception is an explicit custom row whose model id has pinned native Codex capabilities, including Astra or Daybreak on an arbitrary gateway: its advertised list is intersected with diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 2e31c8b2431..3513627da15 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -167,6 +167,8 @@ } }, "explicit": { + "inline-think-boundaries.test.ts": "adapters/openai", + "reasoning-effort-summary-default.test.ts": "responses", "release-desktop-scripts.test.ts": "ci-workflows", "installed-gate-drivers.test.ts": "ci-workflows", "gui-desktop-sidecar-script.test.ts": "gui", @@ -1094,6 +1096,7 @@ "openai-chat-eof.test.ts": "adapters/openai", "openai-chat-hardening.test.ts": "adapters/openai", "openai-chat-image-normalization.test.ts": "adapters/openai", + "openai-chat-inline-think-tags.test.ts": "adapters/openai", "openai-chat-invalid-tool-call-diagnostics.test.ts": "adapters/openai", "openai-chat-model-suffix.test.ts": "adapters/openai", "openai-chat-native-policy.test.ts": "adapters/openai", diff --git a/src/adapters/inline-think-tags.ts b/src/adapters/inline-think-tags.ts new file mode 100644 index 00000000000..2bf7a481740 --- /dev/null +++ b/src/adapters/inline-think-tags.ts @@ -0,0 +1,240 @@ +import type { AdapterEvent } from "../types"; +import { modelInList } from "../types"; +import type { TranslatorBudget } from "../lib/translator-budget"; + +type ThinkingTag = "" | "" | ""; +type ParserState = "pre" | "thinking" | "scanning" | "streaming"; + +const OPEN_TAGS: ThinkingTag[] = ["", "", ""]; +const MAX_OPEN_TAG = Math.max(...OPEN_TAGS.map(t => t.length)); +const MAX_CLOSE_TAG = Math.max(...OPEN_TAGS.map(t => ` tag.startsWith(text) && text.length < tag.length); +} + +/** Move a send boundary back one unit rather than splitting a surrogate pair into U+FFFD. */ +function surrogateSafeCut(text: string, cut: number): number { + if (cut <= 0 || cut >= text.length) return Math.max(0, Math.min(cut, text.length)); + const atCut = text.charCodeAt(cut - 1); + return atCut >= 0xd800 && atCut <= 0xdbff ? cut - 1 : cut; +} + +export interface InlineThinkTagOptions { + /** + * Keep scanning for further think blocks after the first one closes. Kiro emits a single + * leading block, so it leaves this off and streams the rest verbatim. MiniMax M-series + * interleaves several blocks with answer segments, so a reusing adapter opts in. + */ + interleaved?: boolean; +} + +/** + * Recovers thinking that a gateway left inline in visible content as `` blocks instead of + * a separate `reasoning_content` / `reasoning_details` field. Shared by the Kiro adapter and by + * the openai-chat adapter's opt-in `inlineThinkTagModels`. + */ +export class InlineThinkTagParser { + private state: ParserState = "pre"; + private preBuffer = ""; + private thinkingBuffer = ""; + private closeTag = ""; + + private readonly interleaved: boolean; + + constructor(private readonly budget?: TranslatorBudget, options?: InlineThinkTagOptions) { + this.interleaved = options?.interleaved === true; + } + + private replaceCarry(field: "preBuffer" | "thinkingBuffer", next: string): void { + const previous = this[field]; + if (previous === next) return; + const previousBytes = Buffer.byteLength(previous); + const nextBytes = Buffer.byteLength(next); + const reservation = this.budget?.reserveTransient(nextBytes, { kind: "reasoning" }); + this[field] = next; + reservation?.commitRetained(); + this.budget?.releaseRetained(previousBytes, { kind: "reasoning" }); + } + + feed(text: string): AdapterEvent[] { + if (!text) return []; + if (this.state === "streaming") return [{ type: "text_delta", text }]; + if (this.state === "thinking") { + this.replaceCarry("thinkingBuffer", this.thinkingBuffer + text); + return this.drain(); + } + if (this.state === "scanning") { + this.replaceCarry("preBuffer", this.preBuffer + text); + return this.drain(); + } + this.replaceCarry("preBuffer", this.preBuffer + text); + const stripped = this.preBuffer.trimStart(); + const openTag = OPEN_TAGS.find(tag => stripped.startsWith(tag)); + if (openTag) { + const leading = this.interleaved ? this.preBuffer.slice(0, this.preBuffer.length - stripped.length) : ""; + this.state = "thinking"; + this.closeTag = closeTagFor(openTag); + this.replaceCarry("thinkingBuffer", stripped.slice(openTag.length)); + this.replaceCarry("preBuffer", ""); + const events: AdapterEvent[] = leading ? [{ type: "text_delta", text: leading }] : []; + for (const event of this.drain()) events.push(event); + return events; + } + if (stripped.length <= MAX_OPEN_TAG && isPossibleOpenTagPrefix(stripped)) return []; + this.state = "streaming"; + const out = this.preBuffer; + this.replaceCarry("preBuffer", ""); + return out ? [{ type: "text_delta", text: out }] : []; + } + + flush(): AdapterEvent[] { + if (this.state === "thinking") { + const out = this.thinkingBuffer; + this.replaceCarry("thinkingBuffer", ""); + this.state = "streaming"; + return out ? [{ type: "reasoning_raw_delta", text: out }] : []; + } + if (this.preBuffer) { + const out = this.preBuffer; + this.replaceCarry("preBuffer", ""); + this.state = "streaming"; + return [{ type: "text_delta", text: out }]; + } + return []; + } + + /** Release any partial tag/content carry when the owning stream stops early. */ + dispose(): void { + this.replaceCarry("preBuffer", ""); + this.replaceCarry("thinkingBuffer", ""); + this.closeTag = ""; + this.state = "streaming"; + } + + private drain(): AdapterEvent[] { + const events: AdapterEvent[] = []; + // State transitions consume a complete tag; incomplete carry ends this feed. + // Do not recurse for each block in one upstream chunk. + for (;;) { + const before = this.state; + const next = before === "thinking" ? this.drainThinking() : this.drainScanning(); + for (const event of next) events.push(event); + if (this.state === before || this.state === "streaming") return events; + } + } + + private drainThinking(): AdapterEvent[] { + const close = this.closeTag; + const idx = this.thinkingBuffer.indexOf(close); + if (idx >= 0) { + const thinking = this.thinkingBuffer.slice(0, idx); + const remainder = this.thinkingBuffer.slice(idx + close.length); + // Opt-in Chat answers are byte-preserving; keep Kiro's legacy normalization. + const after = this.interleaved ? remainder : remainder.trimStart(); + this.replaceCarry("thinkingBuffer", ""); + const events: AdapterEvent[] = []; + if (thinking) events.push({ type: "reasoning_raw_delta", text: thinking }); + if (this.interleaved) { + this.state = "scanning"; + this.replaceCarry("preBuffer", after); + } else { + this.state = "streaming"; + if (after) events.push({ type: "text_delta", text: after }); + } + return events; + } + if (this.thinkingBuffer.length <= MAX_CLOSE_TAG) return []; + // Hold back a possible partial close tag, and never split a surrogate pair + // at the send boundary: a lone high surrogate encodes as U+FFFD. + const cut = surrogateSafeCut(this.thinkingBuffer, this.thinkingBuffer.length - MAX_CLOSE_TAG); + const send = this.thinkingBuffer.slice(0, cut); + this.replaceCarry("thinkingBuffer", this.thinkingBuffer.slice(cut)); + return send ? [{ type: "reasoning_raw_delta", text: send }] : []; + } + + /** + * Interleaved mode only: the response already proved it carries inline thinking, so a later + * block can open anywhere in the answer text rather than only at the start. + */ + private drainScanning(): AdapterEvent[] { + const events: AdapterEvent[] = []; + let openIndex = -1; + let openTag: ThinkingTag | undefined; + for (const tag of OPEN_TAGS) { + const index = this.preBuffer.indexOf(tag); + if (index >= 0 && (openIndex < 0 || index < openIndex)) { + openIndex = index; + openTag = tag; + } + } + if (openIndex >= 0 && openTag) { + const before = this.preBuffer.slice(0, openIndex); + if (before) events.push({ type: "text_delta", text: before }); + this.state = "thinking"; + this.closeTag = closeTagFor(openTag); + this.replaceCarry("thinkingBuffer", this.preBuffer.slice(openIndex + openTag.length)); + this.replaceCarry("preBuffer", ""); + return events; + } + // Hold back only as much as a partial open tag could occupy. + const cut = surrogateSafeCut(this.preBuffer, this.preBuffer.length - (MAX_OPEN_TAG - 1)); + if (cut > 0) { + events.push({ type: "text_delta", text: this.preBuffer.slice(0, cut) }); + this.replaceCarry("preBuffer", this.preBuffer.slice(cut)); + } + return events; + } +} + +/** Visible-content splitter the openai-chat adapter holds for the life of one response. */ +export interface InlineThinkContentSplitter { + feed(text: string): AdapterEvent[]; + flush(): AdapterEvent[]; + dispose(): void; +} + +const PASSTHROUGH: InlineThinkContentSplitter = { + feed: text => [{ type: "text_delta", text }], + flush: () => [], + dispose: () => { /* nothing carried */ }, +}; + +/** + * Opt-in recovery for `inlineThinkTagModels`. A model that is not listed gets a passthrough that + * never inspects or rewrites visible content, so the 66 registry providers sharing the openai-chat + * adapter keep byte-exact behavior. + */ +export function createInlineThinkContentSplitter( + models: string[] | undefined, + modelId: string | undefined, + budget?: TranslatorBudget, +): InlineThinkContentSplitter { + if (!modelInList(models, modelId ?? "")) return PASSTHROUGH; + const parser = new InlineThinkTagParser(budget, { interleaved: true }); + return { + // An empty content delta stays an empty delta: it is a wire signal, not thinking. + feed: text => (text.length === 0 ? [{ type: "text_delta", text }] : parser.feed(text)), + flush: () => parser.flush(), + dispose: () => parser.dispose(), + }; +} + +/** One-shot form for a non-streaming response body. */ +export function splitInlineThinkContent( + models: string[] | undefined, + modelId: string | undefined, + budget: TranslatorBudget | undefined, + content: string, +): AdapterEvent[] { + const splitter = createInlineThinkContentSplitter(models, modelId, budget); + try { + return [...splitter.feed(content), ...splitter.flush()]; + } finally { + splitter.dispose(); + } +} diff --git a/src/adapters/kiro-thinking.ts b/src/adapters/kiro-thinking.ts deleted file mode 100644 index ee144e28783..00000000000 --- a/src/adapters/kiro-thinking.ts +++ /dev/null @@ -1,112 +0,0 @@ -import type { AdapterEvent } from "../types"; -import type { TranslatorBudget } from "../lib/translator-budget"; - -type ThinkingTag = "" | "" | ""; -type ParserState = "pre" | "thinking" | "streaming"; - -const OPEN_TAGS: ThinkingTag[] = ["", "", ""]; -const MAX_OPEN_TAG = Math.max(...OPEN_TAGS.map(t => t.length)); -const MAX_CLOSE_TAG = Math.max(...OPEN_TAGS.map(t => ` tag.startsWith(text) && text.length < tag.length); -} - -export class KiroThinkingParser { - private state: ParserState = "pre"; - private preBuffer = ""; - private thinkingBuffer = ""; - private closeTag = ""; - - constructor(private readonly budget?: TranslatorBudget) {} - - private replaceCarry(field: "preBuffer" | "thinkingBuffer", next: string): void { - const previous = this[field]; - if (previous === next) return; - const previousBytes = Buffer.byteLength(previous); - const nextBytes = Buffer.byteLength(next); - const reservation = this.budget?.reserveTransient(nextBytes, { kind: "reasoning" }); - this[field] = next; - reservation?.commitRetained(); - this.budget?.releaseRetained(previousBytes, { kind: "reasoning" }); - } - - feed(text: string): AdapterEvent[] { - if (!text) return []; - if (this.state === "streaming") return [{ type: "text_delta", text }]; - if (this.state === "thinking") { - this.replaceCarry("thinkingBuffer", this.thinkingBuffer + text); - return this.drainThinking(); - } - this.replaceCarry("preBuffer", this.preBuffer + text); - const stripped = this.preBuffer.trimStart(); - const openTag = OPEN_TAGS.find(tag => stripped.startsWith(tag)); - if (openTag) { - this.state = "thinking"; - this.closeTag = closeTagFor(openTag); - this.replaceCarry("thinkingBuffer", stripped.slice(openTag.length)); - this.replaceCarry("preBuffer", ""); - return this.drainThinking(); - } - if (stripped.length <= MAX_OPEN_TAG && isPossibleOpenTagPrefix(stripped)) return []; - this.state = "streaming"; - const out = this.preBuffer; - this.replaceCarry("preBuffer", ""); - return out ? [{ type: "text_delta", text: out }] : []; - } - - flush(): AdapterEvent[] { - if (this.state === "thinking") { - const out = this.thinkingBuffer; - this.replaceCarry("thinkingBuffer", ""); - this.state = "streaming"; - return out ? [{ type: "reasoning_raw_delta", text: out }] : []; - } - if (this.preBuffer) { - const out = this.preBuffer; - this.replaceCarry("preBuffer", ""); - this.state = "streaming"; - return [{ type: "text_delta", text: out }]; - } - return []; - } - - /** Release any partial tag/content carry when the owning stream stops early. */ - dispose(): void { - this.replaceCarry("preBuffer", ""); - this.replaceCarry("thinkingBuffer", ""); - this.closeTag = ""; - this.state = "streaming"; - } - - private drainThinking(): AdapterEvent[] { - const close = this.closeTag; - const idx = this.thinkingBuffer.indexOf(close); - if (idx >= 0) { - const thinking = this.thinkingBuffer.slice(0, idx); - const after = this.thinkingBuffer.slice(idx + close.length).trimStart(); - this.replaceCarry("thinkingBuffer", ""); - this.state = "streaming"; - const events: AdapterEvent[] = []; - if (thinking) events.push({ type: "reasoning_raw_delta", text: thinking }); - if (after) events.push({ type: "text_delta", text: after }); - return events; - } - if (this.thinkingBuffer.length <= MAX_CLOSE_TAG) return []; - // Never split a surrogate pair at the send boundary: a lone high - // surrogate at the end of one delta encodes as U+FFFD. Move the cut one - // unit earlier so the whole pair stays in the carry. - let cut = this.thinkingBuffer.length - MAX_CLOSE_TAG; - if (cut > 0 && cut < this.thinkingBuffer.length) { - const atCut = this.thinkingBuffer.charCodeAt(cut - 1); - if (atCut >= 0xd800 && atCut <= 0xdbff) cut -= 1; - } - const send = this.thinkingBuffer.slice(0, cut); - this.replaceCarry("thinkingBuffer", this.thinkingBuffer.slice(cut)); - return send ? [{ type: "reasoning_raw_delta", text: send }] : []; - } -} diff --git a/src/adapters/kiro/stream.ts b/src/adapters/kiro/stream.ts index d10ab1105c0..0536038e7dd 100644 --- a/src/adapters/kiro/stream.ts +++ b/src/adapters/kiro/stream.ts @@ -16,7 +16,7 @@ import { } from "../kiro-errors"; import { parseKiroEvent } from "../kiro-events"; import { noteKiroTransientThrottle } from "../kiro-retry"; -import { KiroThinkingParser } from "../kiro-thinking"; +import { InlineThinkTagParser } from "../inline-think-tags"; import { isCompleteKiroToolInput, kiroTruncationErrorMessage } from "../kiro-truncation"; import { isValidKiroConversationId } from "../kiro-wire"; import { tagKiroReasoningBlob } from "./reasoning"; @@ -319,7 +319,7 @@ async function* parseKiroAttemptEvents( let authoritativeUsage: OcxUsage | undefined; let stopReason: string | undefined; const fallbackEvents: AdapterEvent[] = []; - const thinking = new KiroThinkingParser(budget); + const thinking = new InlineThinkTagParser(budget); const retainedEventBytes = (event: AdapterEvent): number => Buffer.byteLength(JSON.stringify(event)); const retainEvent = (event: AdapterEvent): void => { diff --git a/src/adapters/openai-chat.ts b/src/adapters/openai-chat.ts index 98ba27cd203..dbc1ce76954 100644 --- a/src/adapters/openai-chat.ts +++ b/src/adapters/openai-chat.ts @@ -4,6 +4,7 @@ import { applyExplicitChatReasoningWirePolicy } from "./openai-chat/reasoning-wi import type { AdapterRequest, IncomingMeta, ProviderAdapter } from "./base"; import type { AdapterEvent, OcxParsedRequest, OcxProviderConfig, OcxUsage } from "../types"; import { modelInList } from "../types"; +import { createInlineThinkContentSplitter, splitInlineThinkContent } from "./inline-think-tags"; import { mapReasoningEffort, modelRecordValue } from "../reasoning-effort"; import { debugProviderDiagnostic } from "../lib/debug"; import { sseFieldValue } from "../lib/sse-decoder"; @@ -355,6 +356,12 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd // Gate on the routed model, not list length: a mixed openai-chat provider // can list MiniMax ids without putting every sibling on MiniMax semantics. const reasoningDetailsOptIn = modelInList(provider.reasoningDetailsModels, lastRequestedModelId ?? ""); + // A gateway with no server-side reasoning parser leaves thinking inline in `content` as + // blocks, which would otherwise render as the answer. Passthrough unless opted in. + const inlineThink = createInlineThinkContentSplitter(provider.inlineThinkTagModels, lastRequestedModelId, budget); + const emitContent = function* (events: AdapterEvent[]): Generator { + for (const event of events) { if (event.type === "text_delta") sawUserFacingOutput = true; yield event; } + }; const handleDataLine = function* (line: string): Generator { const rawPayload = sseFieldValue(line, "data"); @@ -362,6 +369,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd const payload = rawPayload.trim(); if (payload.length === 0) return "continue"; if (payload === "[DONE]") { + yield* emitContent(inlineThink.flush()); if ((yield* flushToolCalls()) === "terminate") return "terminate"; const stopReason = stopReasonFor(finishReason); yield { type: "done", usage: pendingUsage, ...(stopReason ? { stopReason } : {}) }; @@ -422,8 +430,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd if (reasoningText !== undefined) yield { type: "reasoning_raw_delta", text: reasoningText }; } if (typeof delta.content === "string" && delta.content.length > 0) { - sawUserFacingOutput = true; - yield { type: "text_delta", text: delta.content }; + yield* emitContent(inlineThink.feed(delta.content)); } const rawToolCalls = delta.tool_calls; @@ -562,6 +569,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd } if (typeof choice.finish_reason === "string" && choice.finish_reason) { + yield* emitContent(inlineThink.flush()); if ((yield* flushToolCalls()) === "terminate") return "terminate"; } return "continue"; @@ -605,6 +613,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd if (buffer.length > 0) { if ((yield* handleDataLine(buffer)) === "terminate") return; } + yield* emitContent(inlineThink.flush()); const sawFinish = finishReason !== undefined; if (!sawFinish && pendingToolCalls.length > 0) { // Some OpenAI-compatible gateways close immediately after a complete function-call @@ -652,6 +661,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd } finally { budget.releaseRetained(bufferBytes, { kind: "live_transient" }); reasoningDetailTracker.release(); + inlineThink.dispose(); closeToolCalls(); reader.releaseLock(); } @@ -738,7 +748,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd if (segments.length > 0) reasoningText = segments.map(s => s.text).join(""); } if (reasoningText !== undefined) events.push({ type: "reasoning_raw_delta", text: reasoningText }); - if (typeof msg.content === "string") events.push({ type: "text_delta", text: msg.content }); + if (typeof msg.content === "string") events.push(...splitInlineThinkContent(provider.inlineThinkTagModels, lastRequestedModelId, budget, msg.content)); const rawToolCalls = msg.tool_calls; if (rawToolCalls !== undefined && rawToolCalls !== null) { if (!Array.isArray(rawToolCalls)) { diff --git a/src/adapters/openai-chat/messages.ts b/src/adapters/openai-chat/messages.ts index 9ad0dec86f8..a712c83158a 100644 --- a/src/adapters/openai-chat/messages.ts +++ b/src/adapters/openai-chat/messages.ts @@ -224,7 +224,7 @@ export function messagesToChatFormat(parsed: OcxParsedRequest, provider: OcxProv let reasoningContent = thinkingParts.map(p => p.thinking).join(""); if ( reasoningContent.length === 0 - && toolCalls.length > 0 + && (toolCalls.length > 0 || thinkingParts.length > 0) && modelInList(provider.preserveReasoningContentModels, parsed.modelId) ) { const cached = toolCalls @@ -235,11 +235,11 @@ export function messagesToChatFormat(parsed: OcxParsedRequest, provider: OcxProv if (cached.length > 0) { reasoningContent = [...new Set(cached)].join("\n"); } else if (modelInList(provider.requiresReasoningPlaceholderModels ?? provider.preserveReasoningContentModels, parsed.modelId)) { - // Fallback (extends #950, closes #1193): the replay cache is + // Fallback (extends #950 and #1193; fixes #5421): the replay cache is // bounded (64 entries / 256 KiB / 1 h TTL) and always misses on - // long sessions, and some tool rounds carry no recorded reasoning - // at all. DeepSeek thinking mode rejects ANY tool_call assistant - // message missing reasoning_content with HTTP 400, so inject a + // long sessions, and some thinking/tool rounds carry no recorded + // reasoning at all. DeepSeek thinking mode rejects replay without + // reasoning_content with HTTP 400, so inject a // minimal placeholder rather than emit a bare continuation the // upstream will reject. Scoped to requiresReasoningPlaceholderModels // (defaulting to the preserve list): preserve-listed providers with diff --git a/src/combos/request.ts b/src/combos/request.ts index e0fa6426087..0d0b0123732 100644 --- a/src/combos/request.ts +++ b/src/combos/request.ts @@ -110,9 +110,13 @@ export function concreteComboRequestBody( return clone; } if (reasoning === undefined) { - clone.reasoning = { effort: resolvedEffort }; + clone.reasoning = { effort: resolvedEffort, summary: "auto" }; } else { - clone.reasoning = { ...(reasoning as Record), effort: resolvedEffort }; + clone.reasoning = { + ...(reasoning as Record), + effort: resolvedEffort, + ...((reasoning as Record).summary === undefined ? { summary: "auto" } : {}), + }; } return clone; } diff --git a/src/providers/derive.ts b/src/providers/derive.ts index 299b95cc413..cc60e657b8a 100644 --- a/src/providers/derive.ts +++ b/src/providers/derive.ts @@ -47,6 +47,7 @@ export interface DerivedKeyLoginProvider { requiresReasoningPlaceholderModels?: string[]; showThinkingSummary?: boolean; reasoningSplitModels?: string[]; + inlineThinkTagModels?: string[]; reasoningDetailsModels?: string[]; thinkingToggleModels?: string[]; thinkingBudgetModels?: string[]; @@ -286,6 +287,7 @@ export function providerConfigSeed(entry: ProviderRegistryEntry): OcxProviderCon ...(entry.requiresReasoningPlaceholderModels ? { requiresReasoningPlaceholderModels: [...entry.requiresReasoningPlaceholderModels] } : {}), ...(entry.showThinkingSummary !== undefined ? { showThinkingSummary: entry.showThinkingSummary } : {}), ...(entry.reasoningSplitModels ? { reasoningSplitModels: [...entry.reasoningSplitModels] } : {}), + ...(entry.inlineThinkTagModels ? { inlineThinkTagModels: [...entry.inlineThinkTagModels] } : {}), ...(entry.reasoningDetailsModels ? { reasoningDetailsModels: [...entry.reasoningDetailsModels] } : {}), ...(entry.thinkingToggleModels ? { thinkingToggleModels: [...entry.thinkingToggleModels] } : {}), ...(entry.thinkingBudgetModels ? { thinkingBudgetModels: [...entry.thinkingBudgetModels] } : {}), @@ -336,6 +338,7 @@ export function deriveKeyLoginMap(): Record { ...(entry.requiresReasoningPlaceholderModels ? { requiresReasoningPlaceholderModels: [...entry.requiresReasoningPlaceholderModels] } : {}), ...(entry.showThinkingSummary !== undefined ? { showThinkingSummary: entry.showThinkingSummary } : {}), ...(entry.reasoningSplitModels ? { reasoningSplitModels: [...entry.reasoningSplitModels] } : {}), + ...(entry.inlineThinkTagModels ? { inlineThinkTagModels: [...entry.inlineThinkTagModels] } : {}), ...(entry.reasoningDetailsModels ? { reasoningDetailsModels: [...entry.reasoningDetailsModels] } : {}), ...(entry.thinkingToggleModels ? { thinkingToggleModels: [...entry.thinkingToggleModels] } : {}), ...(entry.thinkingBudgetModels ? { thinkingBudgetModels: [...entry.thinkingBudgetModels] } : {}), @@ -602,6 +605,7 @@ export function enrichProviderFromRegistry(name: string, prov: OcxProviderConfig if (!prov.preserveReasoningContentModels && seed.preserveReasoningContentModels) prov.preserveReasoningContentModels = [...seed.preserveReasoningContentModels]; if (!prov.requiresReasoningPlaceholderModels && seed.requiresReasoningPlaceholderModels) prov.requiresReasoningPlaceholderModels = [...seed.requiresReasoningPlaceholderModels]; if (!prov.reasoningSplitModels && seed.reasoningSplitModels) prov.reasoningSplitModels = [...seed.reasoningSplitModels]; + if (!prov.inlineThinkTagModels && seed.inlineThinkTagModels) prov.inlineThinkTagModels = [...seed.inlineThinkTagModels]; if (!prov.reasoningDetailsModels && seed.reasoningDetailsModels) prov.reasoningDetailsModels = [...seed.reasoningDetailsModels]; if (!prov.thinkingToggleModels && seed.thinkingToggleModels) prov.thinkingToggleModels = [...seed.thinkingToggleModels]; if (!prov.thinkingBudgetModels && seed.thinkingBudgetModels) prov.thinkingBudgetModels = [...seed.thinkingBudgetModels]; diff --git a/src/providers/model-rename-fields.ts b/src/providers/model-rename-fields.ts index 54cb85575ce..b990a426da3 100644 --- a/src/providers/model-rename-fields.ts +++ b/src/providers/model-rename-fields.ts @@ -121,6 +121,7 @@ export const PROVIDER_MODEL_RENAME_ROLES = { transientRetryOn5xx: "none", retryOnReset: "none", reasoningSplitModels: "list", + inlineThinkTagModels: "list", reasoningDetailsModels: "list", thinkingToggleModels: "list", thinkingBudgetModels: "list", diff --git a/src/providers/registry/model-ids.ts b/src/providers/registry/model-ids.ts index 94de3fd7d99..4c02ca57cad 100644 --- a/src/providers/registry/model-ids.ts +++ b/src/providers/registry/model-ids.ts @@ -118,6 +118,7 @@ export const REGISTRY_FIELD_MODEL_ID_ROLES = { requiresReasoningPlaceholderModels: NONE, showThinkingSummary: NONE, reasoningSplitModels: NONE, + inlineThinkTagModels: NONE, reasoningDetailsModels: NONE, thinkingToggleModels: NONE, thinkingBudgetModels: NONE, diff --git a/src/providers/registry/types.ts b/src/providers/registry/types.ts index 0a31facef64..e04931bdb63 100644 --- a/src/providers/registry/types.ts +++ b/src/providers/registry/types.ts @@ -335,6 +335,8 @@ export interface ProviderRegistryEntry { */ showThinkingSummary?: boolean; reasoningSplitModels?: string[]; + /** See OcxProviderConfig.inlineThinkTagModels. */ + inlineThinkTagModels?: string[]; reasoningDetailsModels?: string[]; thinkingToggleModels?: string[]; thinkingBudgetModels?: string[]; @@ -358,6 +360,6 @@ export type ProviderConfigSeed = Pick< | "modelMaxInputTokens" | "defaultMaxOutputTokens" | "modelMaxOutputTokens" | "reasoningEfforts" | "modelReasoningEfforts" | "modelDefaultReasoningEfforts" | "reasoningEffortMap" | "modelReasoningEffortMap" | "reasoningWireFormat" | "noVisionModels" | "noReasoningModels" | "noTemperatureModels" | "noTopPModels" | "noPenaltyModels" - | "autoToolChoiceOnlyModels" | "preserveReasoningContentModels" | "requiresReasoningPlaceholderModels" | "reasoningSplitModels" | "reasoningDetailsModels" | "thinkingToggleModels" | "thinkingBudgetModels" | "escapeBuiltinToolNames" | "openaiChatEofTolerance" | "showThinkingSummary" + | "autoToolChoiceOnlyModels" | "preserveReasoningContentModels" | "requiresReasoningPlaceholderModels" | "reasoningSplitModels" | "inlineThinkTagModels" | "reasoningDetailsModels" | "thinkingToggleModels" | "thinkingBudgetModels" | "escapeBuiltinToolNames" | "openaiChatEofTolerance" | "showThinkingSummary" | "googleMode" | "project" | "location" | "headers" >; diff --git a/src/providers/resolved-model-policy.ts b/src/providers/resolved-model-policy.ts index 7c3553a3d26..956a62d0996 100644 --- a/src/providers/resolved-model-policy.ts +++ b/src/providers/resolved-model-policy.ts @@ -37,7 +37,7 @@ export type StaticProviderPolicyField = | "supportsResponsesCustomTools" | "preserveResponsesReasoningContent" | "dropResponsesReasoningItems" | "modelSupportsReasoningSummaries" | "supportsVerbosity" | "modelSupportsVerbosity" | "responsesItemIdRepair" | "autoToolChoiceOnlyModels" - | "preserveReasoningContentModels" | "requiresReasoningPlaceholderModels" | "reasoningSplitModels" + | "preserveReasoningContentModels" | "requiresReasoningPlaceholderModels" | "reasoningSplitModels" | "inlineThinkTagModels" | "reasoningDetailsModels" | "thinkingToggleModels" | "thinkingBudgetModels" | "showThinkingSummary" | "escapeBuiltinToolNames" | "googleMode" | "project" | "location" | "modelCapabilities" | "modelAutoCompactTokenLimits" | "modelSuppressSyntheticMax" | "modelReasoningSummaryDelivery" @@ -230,6 +230,8 @@ export function resolveModelPolicy(input: ResolveModelPolicyInput): ResolvedMode "preserveReasoningContentModels", "requiresReasoningPlaceholderModels", "reasoningSplitModels", "reasoningDetailsModels", "thinkingToggleModels", "thinkingBudgetModels", ] as const) putUnion(key, entry?.[key]); + // This parser is opt-in: an explicit list, including [], overrides registry defaults. + putScalar("inlineThinkTagModels", entry?.inlineThinkTagModels); for (const directModel of entry?.directReasoningEffortModels ?? []) { const staleBudget = [directModel, ...(entry?.thinkingBudgetModels ?? [])]; const routedStaleBudget = [...(entry?.thinkingBudgetModels ?? []), directModel]; diff --git a/src/responses/parser.ts b/src/responses/parser.ts index 51bb36f5245..914c2e9cae4 100644 --- a/src/responses/parser.ts +++ b/src/responses/parser.ts @@ -542,7 +542,8 @@ export function parseRequest( options.reasoning = requestedEffort; } const summaryMode = data.reasoning?.summary; - if (!summaryMode || summaryMode === "none") options.hideThinkingSummary = true; + const reasoningActive = options.reasoning !== undefined && options.reasoning !== "none"; + if (summaryMode === "none" || (!summaryMode && !reasoningActive)) options.hideThinkingSummary = true; if (data.presence_penalty !== undefined) options.presencePenalty = data.presence_penalty; if (data.frequency_penalty !== undefined) options.frequencyPenalty = data.frequency_penalty; if (data.service_tier !== undefined) options.serviceTier = data.service_tier; diff --git a/src/router.ts b/src/router.ts index 6d5e4b7310b..3e4182dcc15 100644 --- a/src/router.ts +++ b/src/router.ts @@ -337,6 +337,7 @@ export function routedProviderConfig(providerName: string, provider: OcxProvider const preserveReasoningContentModels = staticPolicy.preserveReasoningContentModels; const requiresReasoningPlaceholderModels = staticPolicy.requiresReasoningPlaceholderModels; const reasoningSplitModels = staticPolicy.reasoningSplitModels; + const inlineThinkTagModels = staticPolicy.inlineThinkTagModels; const reasoningDetailsModels = staticPolicy.reasoningDetailsModels; const thinkingToggleModels = staticPolicy.thinkingToggleModels; const thinkingBudgetModels = staticPolicy.thinkingBudgetModels; @@ -479,6 +480,7 @@ export function routedProviderConfig(providerName: string, provider: OcxProvider ...(preserveReasoningContentModels ? { preserveReasoningContentModels } : {}), ...(requiresReasoningPlaceholderModels ? { requiresReasoningPlaceholderModels } : {}), ...(reasoningSplitModels ? { reasoningSplitModels } : {}), + ...(inlineThinkTagModels ? { inlineThinkTagModels } : {}), ...(reasoningDetailsModels ? { reasoningDetailsModels } : {}), ...(thinkingToggleModels ? { thinkingToggleModels } : {}), ...(thinkingBudgetModels ? { thinkingBudgetModels } : {}), diff --git a/src/server/auth-cors.ts b/src/server/auth-cors.ts index a0ef077dd5d..f5d71343199 100644 --- a/src/server/auth-cors.ts +++ b/src/server/auth-cors.ts @@ -1060,6 +1060,7 @@ const PROVIDER_CONFIG_FIELD_POLICY = { transientRetryOn5xx: "editor", retryOnReset: "editor", reasoningSplitModels: "editor", + inlineThinkTagModels: "editor", reasoningDetailsModels: "editor", thinkingToggleModels: "editor", thinkingBudgetModels: "editor", diff --git a/src/types/provider.ts b/src/types/provider.ts index 3dbf793d9fd..a36bf94074e 100644 --- a/src/types/provider.ts +++ b/src/types/provider.ts @@ -950,6 +950,16 @@ export interface OcxProviderConfig { * thinking separately in `reasoning_content` / `reasoning_details` instead of visible content. */ reasoningSplitModels?: string[]; + /** + * Model ids served by a gateway that runs no server-side reasoning parser, so a thinking model + * leaves its chain of thought inline in `content` as `` / `` / `` + * blocks and never sends `reasoning_content` or `reasoning_details`. Without this the whole + * chain of thought renders as the answer. The openai-chat adapter then splits those blocks back + * into reasoning. Off by default and narrow on purpose: 66 registry providers share this + * adapter, and a gateway that does parse reasoning must not have its visible content rewritten. + * Prefer a provider-side parser or `reasoningSplitModels` when the upstream supports either. + */ + inlineThinkTagModels?: string[]; /** * Model ids whose chat endpoint carries thinking as a structured `reasoning_details` array * (MiniMax M-series with `reasoning_split`): stream deltas repeat each detail's `text` as a diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index 5be71c0bfbc..9147b59e3a8 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -766,6 +766,9 @@ its defaults and exclusions are owned by [Responses transport](transports/respon The provider editor field policy exposes `showThinkingSummary` as a boolean provider option; it controls Responses summary defaults without a dashboard rendering change. See [Google provider](providers/google.md). +The same editor policy accepts the per-model `inlineThinkTagModels` string list. Its opt-in +format contract is owned by [Chat compatibility](providers/chat-compat.md#inline-think-tag-recovery). + Paginated and migration-capable history follows the [authoritative writer contract](codex-home.md#paginated-history-writer-boundary); this document adds no independent writer guarantee. Codex pool settings and their consumers follow the [reset-first ordering contract](providers/openai-tiers.md#reset-first-account-ordering), including independent-quota fallback, preserved affinity, strategy-specific threshold summaries, and shared short-observation freshness for switch warnings. Codex account DTOs and cards expose the routing-plan exclusion separately from credential health; the [plan exclusion contract](providers/openai-tiers.md#automatic-pool-plan-exclusions) also governs CLI projection. Private pool credential metadata follows the [quota-history publication identity contract](providers/openai-tiers.md#quota-history-publication-identity); credential-only and account DTO projections omit it. diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index bf4271d54fe..82955de036d 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -1,5 +1,8 @@ # Providers And Adapters +The opt-in `inlineThinkTagModels` list follows static-policy override and model-rename rules; +shared Kiro/Chat splitting and raw display follow [Chat compatibility](providers/chat-compat.md#inline-think-tag-recovery). + OrcaRouter key exchange uses the shared raw-byte reader before returning a durable key. Its 64 KiB response ceiling, single 30-second header/body deadline, and cancellation behavior follow the [bounded ingestion contract](transports/inventory.md#bounded-response-ingestion-and-orcarouter-login). diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index a14c7c24d98..e71e3a67681 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -46,6 +46,32 @@ Shared parsing and streaming follow the [request-copy](../transports/byte-accoun ## Reasoning and tool-result compatibility +### Inline think-tag recovery + +A gateway that serves a thinking model without a server-side reasoning parser returns the chain +of thought inside `message.content` as `` / `` / `` blocks and sends +neither `reasoning_content` nor `reasoning_details`. `src/adapters/openai-chat.ts` recovers those +blocks into reasoning only for models listed in `inlineThinkTagModels`. An explicit operator list, +including `[]`, replaces matching registry defaults. The option is off by default because registry +providers share this adapter and a gateway that does parse reasoning +must keep its visible content byte-exact. Once enabled the splitter still engages only for a +response that opens with a thinking tag (optionally preceded by whitespace), so ordinary prose or +code fences before a tag leave the entire response untouched. Whitespace before that initial tag +and after every closing tag remains answer text; +after it engages it keeps splitting later blocks, because M-series models interleave thinking with +answer segments, including same-line interleaving. Once engaged, tags are protocol delimiters even +inside subsequent code fences or quoted examples: this explicit opt-in does not parse Markdown. +Gateways producing ambiguous literals should use structured reasoning instead. Iterative draining +keeps stack depth independent of the number of blocks in an upstream chunk. +A block left unterminated at end of stream flushes as reasoning rather than being +dropped. Whitespace between the leading block and the start of the answer is dropped as formatting +noise; once answer text has been emitted, whitespace after a later closing tag is preserved, +because a mid-answer block sits inside markdown or code where indentation is meaningful. +`src/adapters/inline-think-tags.ts` owns the parser and is shared with the Kiro adapter, +which consumes it in single-block mode with its existing leading/first-answer normalization. +Regression coverage is in `tests/adapters/openai/openai-chat-inline-think-tags.test.ts` and +`tests/adapters/openai/inline-think-boundaries.test.ts`. + Google tool-declaration narrowing is observed by the Google final compiler, not this shared Chat compatibility layer. Its endpoint profile and privacy boundary are specified in the [Google provider contract](google.md#google-tool-schema-loss-reporting). @@ -362,7 +388,10 @@ measurement rather than allocating a serialized copy just to measure it. > Decision record: [ADR-0067](../decisions/ADR-0067-reasoning-display-parity-hidethinkingsummary.md) -`hideThinkingSummary` (request reasoning summary absent/"none" — the routed catalog default) is +`hideThinkingSummary` is set for explicit summary "none", or omitted summary without a validated +active effort. Accepted minimal/low/medium/high/xhigh/max (including ultra normalized to max) +allow raw visibility when summary is omitted; none and invalid efforts do not. Explicit "auto" +still permits raw visibility independently of effort. This flag is honored by BOTH reasoning paths: anthropic `thinking_delta` AND raw `reasoning_raw_delta` (openai-chat `reasoning_content`, kiro tags). Hidden reasoning emits an envelope-only reasoning item (`summary: []`, txt-only `ocxr1:` `encrypted_content`, no text deltas) — invisible in the @@ -376,6 +405,10 @@ the desktop thinking band shows the "Thinking…" placeholder, and raw text appe which only fits native OpenAI providers that author real summaries. Diagnosis and codex-rs grouping evidence: `devlog/_fin/260709_native_response_pattern/`. +For models that require a reasoning placeholder, a preserved thinking-only assistant turn with no +plaintext receives that placeholder even when it has no tool call. Otherwise the Chat serializer +drops the turn and strict DeepSeek continuations can reject the following request (#5421). + The process-local raw-reasoning fallback is fail-closed unless a request has an explicit client thread plus an exact provider destination, wire adapter, final model, and physical credential identity. API-key material is represented only by a process-keyed HMAC; OAuth replay is bound to the diff --git a/structure/transports/responses.md b/structure/transports/responses.md index 023bddbe114..20d5871f9b3 100644 --- a/structure/transports/responses.md +++ b/structure/transports/responses.md @@ -968,6 +968,10 @@ removes those controls for unknown ladders and preserves `reasoning.summary`. Kn ladders retain the existing per-target effort resolution. This request normalization does not change target order or attempt accounting; provider-400 decisions follow the [request-local target compatibility](../runtime.md#request-local-target-compatibility) contract. +An injected combo default supplies `summary: "auto"` only when no summary was specified; caller +summary choices remain intact. Raw display and hidden-envelope replay follow +[reasoning display parity](../providers/chat-compat.md#reasoning-display-parity-hidethinkingsummary). + The shared Responses path follows the [bounded multipart recovery contract](../subagents.md#multipart-encrypted-task-recovery); credential admission and retry policy remain unchanged. ## Upstream key attempt accounting diff --git a/tests/adapters/openai/inline-think-boundaries.test.ts b/tests/adapters/openai/inline-think-boundaries.test.ts new file mode 100644 index 00000000000..4bda09221d0 --- /dev/null +++ b/tests/adapters/openai/inline-think-boundaries.test.ts @@ -0,0 +1,67 @@ +import { describe, expect, test } from "bun:test"; +import { InlineThinkTagParser, splitInlineThinkContent } from "../../../src/adapters/inline-think-tags"; +import type { AdapterEvent } from "../../../src/types"; +import { createTestTranslatorBudget } from "../../helpers/translator-budget"; + +function projection(events: AdapterEvent[]) { + return { + answer: events.filter(e => e.type === "text_delta").map(e => e.text).join(""), + reasoning: events.filter(e => e.type === "reasoning_raw_delta").map(e => e.text).join(""), + }; +} +function split(chunks: string[], interleaved = true) { + const parser = new InlineThinkTagParser(undefined, { interleaved }); + const events: AdapterEvent[] = []; + try { + for (const chunk of chunks) for (const event of parser.feed(chunk)) events.push(event); + events.push(...parser.flush()); + return projection(events); + } finally { parser.dispose(); } +} + +describe("inline thinking format boundaries", () => { + test("leading and first-answer whitespace survives every chunk boundary", () => { + const input = " \nwhy😀\n code();next tail"; + const expected = { answer: " \n\n code(); tail", reasoning: "why😀next" }; + for (let cut = 0; cut <= input.length; cut++) { + expect(split([input.slice(0, cut), input.slice(cut)])).toEqual(expected); + } + }); + test("Kiro retains single-block normalization and subsequent literal tags", () => { + expect(split([" \nwhy\n answerliteral"], false)) + .toEqual({ answer: "answerliteral", reasoning: "why" }); + }); + test("ordinary code examples never activate parsing", () => { + const input = "```xml\nliteral\n```"; + expect(split([...input])).toEqual({ answer: input, reasoning: "" }); + }); + test("after activation tags remain format delimiters even inside a code fence", () => { + const input = "first```xml\nlater\n```"; + expect(split([...input])).toEqual({ answer: "```xml\n\n```", reasoning: "firstlater" }); + }); + test("many same-chunk blocks and split chunks have identical output without recursion", () => { + const block = "ra"; + const count = 12000; + const expected = { answer: "a".repeat(count), reasoning: "r".repeat(count) }; + expect(split([block.repeat(count)])).toEqual(expected); + expect(split(Array(count).fill(block))).toEqual(expected); + }); + test("partial tags and unterminated reasoning flush without loss", () => { + expect(split(["why😀xanswer { + const content = "whyanswer"; + expect(projection(splitInlineThinkContent(["other"], "model", undefined, content))) + .toEqual({ answer: content, reasoning: "" }); + }); + test("dispose releases partial carry and an overflow does not relax the budget", () => { + const budget = createTestTranslatorBudget({ maxTurnBytes: 128 }); + const parser = new InlineThinkTagParser(budget, { interleaved: true }); + parser.feed("partial"); + expect(() => parser.feed("x".repeat(129))).toThrow(); + parser.dispose(); + expect(budget.snapshot().currentBytes).toBe(0); + }); +}); diff --git a/tests/adapters/openai/openai-chat-inline-think-tags.test.ts b/tests/adapters/openai/openai-chat-inline-think-tags.test.ts new file mode 100644 index 00000000000..d2bd917f60c --- /dev/null +++ b/tests/adapters/openai/openai-chat-inline-think-tags.test.ts @@ -0,0 +1,150 @@ +import { describe, expect, test } from "bun:test"; +import { createOpenAIChatAdapter as createOpenAIChatAdapterProduction } from "../../../src/adapters/openai-chat"; +import type { AdapterEvent, OcxParsedRequest, OcxProviderConfig } from "../../../src/types"; +import { withTestTranslatorBudget } from "../../helpers/translator-budget"; +import { buildResponseJSON } from "../../../src/bridge"; +import { parseRequest } from "../../../src/responses/parser"; +import { decodeReasoningEnvelope } from "../../../src/responses/reasoning-envelope"; + +const MODEL = "GLM-5.3-Flash"; + +function provider(optIn: boolean): OcxProviderConfig { + return { + adapter: "openai-chat", + baseUrl: "https://example.test/v1", + apiKey: "key", + ...(optIn ? { inlineThinkTagModels: [MODEL] } : {}), + }; +} + +function parsed(): OcxParsedRequest { + return { + modelId: MODEL, + stream: true, + options: {}, + context: { messages: [{ role: "user", content: "ping", timestamp: 0 }] }, + }; +} + +/** parseStream gates on the routed model, which buildRequest records. */ +function adapterFor(optIn: boolean) { + const adapter = withTestTranslatorBudget(createOpenAIChatAdapterProduction(provider(optIn))); + adapter.buildRequest(parsed()); + return adapter; +} + +function sse(...contents: string[]): Response { + const lines = contents.map(text => `data: ${JSON.stringify({ choices: [{ delta: { content: text } }] })}\n\n`); + lines.push('data: {"choices":[{"delta":{},"finish_reason":"stop"}]}\n\n', "data: [DONE]\n\n"); + return new Response(lines.join("")); +} + +async function collect(gen: AsyncGenerator): Promise { + const out: AdapterEvent[] = []; + for await (const event of gen) if (event.type !== "heartbeat") out.push(event); + return out; +} + +function joined(events: AdapterEvent[], type: "text_delta" | "reasoning_raw_delta"): string { + return events + .filter((event): event is Extract => event.type === type) + .map(event => event.text) + .join(""); +} + +describe("openai-chat inline recovery", () => { + test.each(["none", undefined])("inline reasoning crosses display and next-turn replay with summary %s", async summary => { + const request = parseRequest({ model: MODEL, input: [], reasoning: { effort: "high", summary } }); + const events = await collect(adapterFor(true).parseStream(sse("replay meanswer"))); + const response = buildResponseJSON(events, MODEL, { hideThinkingSummary: request.options.hideThinkingSummary }); + const output = (response as { output: Record[] }).output; + const item = output.find(o => o.type === "reasoning")!; + expect(item.summary).toEqual([]); + if (summary === "none") { + expect(item.content).toBeUndefined(); + expect(decodeReasoningEnvelope(item.encrypted_content as string)?.txt).toBe("replay me"); + } else { + expect(item.content).toEqual([{ type: "reasoning_text", text: "replay me" }]); + } + const next = parseRequest({ model: MODEL, input: [...output, { type: "message", role: "user", content: "next" }] }); + const replayAdapter = withTestTranslatorBudget(createOpenAIChatAdapterProduction({ + ...provider(true), preserveReasoningContentModels: [MODEL], + })); + const wire = JSON.parse(replayAdapter.buildRequest(next).body); + expect(wire.messages.find((m: { role: string }) => m.role === "assistant").reasoning_content).toBe("replay me"); + }); + test("a gateway without a reasoning parser has its thinking split out of the answer", async () => { + const events = await collect(adapterFor(true).parseStream( + sse("weigh", "ing it up", "OCX_THINK_OK"), + )); + + expect(joined(events, "reasoning_raw_delta")).toBe("weighing it up"); + expect(joined(events, "text_delta")).toBe("OCX_THINK_OK"); + expect(events.at(-1)?.type).toBe("done"); + }); + + test("a tag split across chunk boundaries is still recognized", async () => { + const events = await collect(adapterFor(true).parseStream( + sse("whyanswer"), + )); + + expect(joined(events, "reasoning_raw_delta")).toBe("why"); + expect(joined(events, "text_delta")).toBe("answer"); + }); + + test("interleaved blocks keep every later thought out of the answer", async () => { + const events = await collect(adapterFor(true).parseStream( + sse("first", "part one ", "second", "part two"), + )); + + expect(joined(events, "reasoning_raw_delta")).toBe("firstsecond"); + expect(joined(events, "text_delta")).toBe("part one part two"); + }); + + test("a response that opens with ordinary text is never rewritten", async () => { + const answer = "A model may discuss a tag without thinking in it."; + const events = await collect(adapterFor(true).parseStream(sse(answer))); + + expect(joined(events, "text_delta")).toBe(answer); + expect(events.some(event => event.type === "reasoning_raw_delta")).toBe(false); + }); + + test("the initial blank line and the answer's own indentation both survive", async () => { + const events = await collect(adapterFor(true).parseStream( + sse("plan\n\nUse this:\n", "check", " indented line"), + )); + + expect(joined(events, "reasoning_raw_delta")).toBe("plancheck"); + expect(joined(events, "text_delta")).toBe("\n\nUse this:\n indented line"); + }); + + test("an unterminated block is flushed as reasoning rather than lost", async () => { + const events = await collect(adapterFor(true).parseStream( + sse("cut off mid thought"), + )); + + expect(joined(events, "reasoning_raw_delta")).toBe("cut off mid thought"); + expect(joined(events, "text_delta")).toBe(""); + }); + + test("without the opt-in the same stream stays byte-exact visible content", async () => { + const events = await collect(adapterFor(false).parseStream( + sse("weighing it up", "OCX_THINK_OK"), + )); + + expect(joined(events, "text_delta")).toBe("weighing it upOCX_THINK_OK"); + expect(events.some(event => event.type === "reasoning_raw_delta")).toBe(false); + }); + + test("the non-streaming path splits the same way", async () => { + const adapter = adapterFor(true); + const response = new Response(JSON.stringify({ + choices: [{ message: { role: "assistant", content: "quietlyOCX_THINK_OK" } }], + }), { headers: { "content-type": "application/json" } }); + + const events = await adapter.parseResponse(response); + + expect(joined(events, "reasoning_raw_delta")).toBe("quietly"); + expect(joined(events, "text_delta")).toBe("OCX_THINK_OK"); + }); +}); diff --git a/tests/codex-integration/combos.test.ts b/tests/codex-integration/combos.test.ts index b2fc6d3876f..00cbf4a28fd 100644 --- a/tests/codex-integration/combos.test.ts +++ b/tests/codex-integration/combos.test.ts @@ -291,7 +291,7 @@ describe("combo request cloning", () => { expect(concrete).toEqual({ model: "a/m1", input: [{ role: "user", content: "hi" }], - reasoning: { effort: "high" }, + reasoning: { effort: "high", summary: "auto" }, }); expect(raw).toEqual({ model: "combo/free", input: [{ role: "user", content: "hi" }] }); expect(concrete.input).not.toBe(raw.input); @@ -430,15 +430,15 @@ describe("combo request cloning", () => { */ test("a combo default above the target ladder is downgraded, not dropped (#3108)", () => { expect(concreteComboRequestBody({ model: "combo/x" }, target, "max", ["low", "medium", "high"]).reasoning) - .toEqual({ effort: "high" }); + .toEqual({ effort: "high", summary: "auto" }); expect(concreteComboRequestBody({ model: "combo/x" }, target, "high", ["low", "medium"]).reasoning) - .toEqual({ effort: "medium" }); + .toEqual({ effort: "medium", summary: "auto" }); // Exact support is still passed through untouched. expect(concreteComboRequestBody({ model: "combo/x" }, target, "max", ["high", "max"]).reasoning) - .toEqual({ effort: "max" }); + .toEqual({ effort: "max", summary: "auto" }); // Never raises: a request below everything supported takes the lowest rung, not a higher one. expect(concreteComboRequestBody({ model: "combo/x" }, target, "low", ["high", "max"]).reasoning) - .toEqual({ effort: "high" }); + .toEqual({ effort: "high", summary: "auto" }); // A caller-supplied effort still wins over the combo default. expect(concreteComboRequestBody( { model: "combo/x", reasoning: { effort: "low" } }, target, "max", ["low", "medium", "high"], diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 25bf5372234..089074d49b2 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1,4 +1,6 @@ { + "inline-think-boundaries.test.ts": "adapters/openai", + "reasoning-effort-summary-default.test.ts": "responses", "release-desktop-scripts.test.ts": "ci-workflows", "installed-gate-drivers.test.ts": "ci-workflows", "gui-desktop-sidecar-script.test.ts": "gui", @@ -920,6 +922,7 @@ "openai-chat-eof.test.ts": "adapters/openai", "openai-chat-hardening.test.ts": "adapters/openai", "openai-chat-image-normalization.test.ts": "adapters/openai", + "openai-chat-inline-think-tags.test.ts": "adapters/openai", "openai-chat-invalid-tool-call-diagnostics.test.ts": "adapters/openai", "openai-chat-model-suffix.test.ts": "adapters/openai", "openai-chat-native-policy.test.ts": "adapters/openai", diff --git a/tests/providers/deepseek-reasoning-replay-gaps.test.ts b/tests/providers/deepseek-reasoning-replay-gaps.test.ts index 4e5c6ea3cae..74a926fb4a1 100644 --- a/tests/providers/deepseek-reasoning-replay-gaps.test.ts +++ b/tests/providers/deepseek-reasoning-replay-gaps.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, test } from "bun:test"; import { createOpenAIChatAdapter } from "../../src/adapters/openai-chat"; import { buildResponseJSON } from "../../src/bridge"; import { parseRequest } from "../../src/responses/parser"; +import { encodeReasoningEnvelope } from "../../src/responses/reasoning-envelope"; import { clearReasoningReplayCacheForTests, peekReasoningForCall as peekReasoningForCallRaw, @@ -129,6 +130,30 @@ describe("issue #950 — tool-call reasoning replay invariant (openai-chat wire) expect(assistant!["reasoning_content"]).toBe(REASONING); }); + test("GAP F (issue #5421): signed thinking-only turn without tools gets a placeholder", () => { + const { messages } = wireFor([ + { + type: "reasoning", + id: "rs_empty", + summary: [], + encrypted_content: encodeReasoningEnvelope({ sig: "opaque-signature" }), + }, + { + type: "agent_message", + author: "parent", + recipient: "child", + content: [{ type: "input_text", text: "return OK" }], + }, + ]); + + const assistantIndex = messages.findIndex(message => message.role === "assistant"); + const taskIndex = messages.findIndex(message => message.role === "user" && message.content === "return OK"); + expect(assistantIndex).toBeGreaterThanOrEqual(0); + expect(assistantIndex).toBeLessThan(taskIndex); + expect(messages[assistantIndex]!["reasoning_content"]).toBe(" "); + expect(messages[assistantIndex]!.content).toBe(""); + }); + test("GAP A: reasoning item arriving AFTER its function_call is attached to its turn", () => { // Reconstructed histories (resume/retry/synthetic) may order the reasoning // item after the call it belongs to. The parser used to clear the pending @@ -287,6 +312,23 @@ describe("issue #950 — tool-call reasoning replay invariant (openai-chat wire) const miss = toolCallAssistant(missResult.wire.messages); expect(miss).toBeDefined(); expect(miss!["reasoning_content"]).toBeUndefined(); + // Signed thinking without plaintext also stays omitted for an explicit + // placeholder opt-out, even though the parser preserves the thinking turn. + const thinkingOnly = minimaxWire([ + { + type: "reasoning", + id: "rs_minimax_empty", + summary: [], + encrypted_content: encodeReasoningEnvelope({ sig: "opaque-signature" }), + }, + { + type: "agent_message", + author: "parent", + recipient: "child", + content: [{ type: "input_text", text: "return OK" }], + }, + ]).wire.messages; + expect(thinkingOnly.some(message => message.role === "assistant")).toBeFalse(); // Cache hit on the same path: the recorded reasoning still replays. rememberReasoningForCall("call_1", REASONING, missResult.replayScope); const hit = toolCallAssistant(minimaxWire([userMessage(), functionCallOutputItem()]).wire.messages); diff --git a/tests/providers/kiro/kiro-stream.test.ts b/tests/providers/kiro/kiro-stream.test.ts index 336f264d021..5ad687ef6f7 100644 --- a/tests/providers/kiro/kiro-stream.test.ts +++ b/tests/providers/kiro/kiro-stream.test.ts @@ -2231,8 +2231,8 @@ describe("kiro adapter — non-streaming parseResponse", () => { describe("surrogate safety at kiro boundaries", () => { test("the reasoning carry never emits a delta ending on a lone high surrogate", async () => { - const { KiroThinkingParser } = await import("../../../src/adapters/kiro-thinking"); - const parser = new KiroThinkingParser(); + const { InlineThinkTagParser } = await import("../../../src/adapters/inline-think-tags"); + const parser = new InlineThinkTagParser(); // An astral char exactly at the carry/send boundary. const events = parser.feed("🎆aaaaaaaaaaa"); const emitted = JSON.stringify(events); diff --git a/tests/providers/model-rename-migration.test.ts b/tests/providers/model-rename-migration.test.ts index 704fb4b7874..924cecd5967 100644 --- a/tests/providers/model-rename-migration.test.ts +++ b/tests/providers/model-rename-migration.test.ts @@ -48,6 +48,7 @@ function staleConfig(): OcxConfig { modelDefaultReasoningEfforts: { "qwen3.8-max-preview": "xhigh" }, preserveReasoningContentModels: ["glm-5.2", "qwen3.8-max-preview", "qwen3.7-max"], thinkingBudgetModels: ["qwen3.8-max-preview", "qwen3.7-max"], + inlineThinkTagModels: ["qwen3.8-max-preview", "qwen3.7-max"], retainModels: ["qwen3.8-max-preview"], }, }, @@ -71,6 +72,7 @@ describe("registry model rename migration (#1610)", () => { expect(prov.modelDefaultReasoningEfforts?.["qwen3.8-max"]).toBe("xhigh"); expect(prov.preserveReasoningContentModels).toEqual(["glm-5.2", "qwen3.8-max", "qwen3.7-max"]); expect(prov.thinkingBudgetModels).toEqual(["qwen3.8-max", "qwen3.7-max"]); + expect(prov.inlineThinkTagModels).toEqual(["qwen3.8-max", "qwen3.7-max"]); expect(prov.retainModels).toEqual(["qwen3.8-max"]); expect(warnings.some(w => w.includes("qwen3.8-max"))).toBe(true); }); @@ -98,6 +100,7 @@ describe("registry model rename migration (#1610)", () => { prov.modelDefaultReasoningEfforts = {}; prov.preserveReasoningContentModels = ["qwen3.8-max"]; prov.thinkingBudgetModels = ["qwen3.7-max"]; + prov.inlineThinkTagModels = ["qwen3.8-max"]; prov.retainModels = ["qwen3.8-max"]; clean.disabledModels = ["other/model"]; diff --git a/tests/providers/resolved-model-policy.test.ts b/tests/providers/resolved-model-policy.test.ts index 99ecc9411bc..77d8029de59 100644 --- a/tests/providers/resolved-model-policy.test.ts +++ b/tests/providers/resolved-model-policy.test.ts @@ -65,6 +65,14 @@ function resolve( } describe("resolved static model policy parity", () => { + test("inline tag opt-in inherits only matching transport defaults and preserves explicit empty lists", () => { + const entry = registry({ inlineThinkTagModels: [MODEL] }); + expect(resolve(provider(), entry).provider.inlineThinkTagModels).toEqual([MODEL]); + expect(resolve(provider({ inlineThinkTagModels: [] }), entry).provider.inlineThinkTagModels).toEqual([]); + expect(resolve(provider({ inlineThinkTagModels: ["other"] }), entry).provider.inlineThinkTagModels).toEqual(["other"]); + expect(resolve(provider(), entry, false).provider.inlineThinkTagModels).toBeUndefined(); + expect(routedProviderConfig("fixture-provider", provider({ inlineThinkTagModels: [MODEL] })).inlineThinkTagModels).toEqual([MODEL]); + }); test("representative registry maps stay byte-equivalent to the current route merge", () => { const entry = PROVIDER_REGISTRY.find(candidate => { if (candidate.allowBaseUrlOverride || /\{[^}]*\}/.test(candidate.baseUrl)) return false; diff --git a/tests/responses/reasoning-effort-summary-default.test.ts b/tests/responses/reasoning-effort-summary-default.test.ts new file mode 100644 index 00000000000..10f57d60dbb --- /dev/null +++ b/tests/responses/reasoning-effort-summary-default.test.ts @@ -0,0 +1,85 @@ +import { describe, expect, test } from "bun:test"; +import { parseRequest } from "../../src/responses/parser"; +import { concreteComboRequestBody } from "../../src/combos/request"; +import type { OcxComboTarget } from "../../src/types"; + +describe("reasoning effort preserves visible thinking when summary is omitted", () => { + test.each(["unknown", "off", ""])("invalid effort %j does not enable visibility", effort => { + const parsed = parseRequest({ model: "test", input: [], reasoning: { effort } }); + expect(parsed.options.reasoning).toBeUndefined(); + expect(parsed.options.hideThinkingSummary).toBe(true); + }); + test.each([{}, [], 1, null].map(effort => ({ effort })))("non-string effort %j is rejected by the wire schema", ({ effort }) => { + expect(() => parseRequest({ model: "test", input: [], reasoning: { effort } })).toThrow("responses parse error"); + }); + test.each(["minimal", "low", "medium", "high", "xhigh", "max", "ultra"])("validated effort %s enables raw visibility", effort => { + const parsed = parseRequest({ model: "test", input: [], reasoning: { effort } }); + expect(parsed.options.reasoning).toBe(effort === "ultra" ? "max" : effort); + expect(parsed.options.hideThinkingSummary).toBeUndefined(); + }); + test.each([undefined, "none", "off", "invalid"])("explicit auto is independent of effort %s", effort => { + expect(parseRequest({ model: "test", input: [], reasoning: { effort, summary: "auto" } }) + .options.hideThinkingSummary).toBeUndefined(); + }); + test("combo preserves caller summaries, none/minimal, fallback, and adaptive boundaries", () => { + const target = { provider: "test", model: "model" }; + for (const summary of ["none", "auto"]) { + expect(concreteComboRequestBody({ reasoning: { summary } }, target, "high", ["high"]).reasoning) + .toEqual({ effort: "high", summary }); + } + for (const effort of ["none", "minimal", "medium"]) { + expect(concreteComboRequestBody({ reasoning: { effort } }, target, "high", ["high"]).reasoning) + .toEqual({ effort }); + } + expect(concreteComboRequestBody({ reasoning: { effort: "medium", summary: "none" } }, target, "high", ["high"], "strict", "force").reasoning) + .toEqual({ effort: "high", summary: "none" }); + expect(concreteComboRequestBody({ reasoning: { effort: "high", summary: "none" } }, target, "high", undefined, "adaptive").reasoning) + .toEqual({ summary: "none" }); + }); + test("reasoning with active effort does not default to hideThinkingSummary", () => { + const parsed = parseRequest({ + model: "test-model", + reasoning: { effort: "high" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + }); + expect(parsed.options.reasoning).toBe("high"); + expect(parsed.options.hideThinkingSummary).toBeUndefined(); + }); + + test("explicit summary of none still hides thinking summary", () => { + const parsed = parseRequest({ + model: "test-model", + reasoning: { effort: "high", summary: "none" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + }); + expect(parsed.options.reasoning).toBe("high"); + expect(parsed.options.hideThinkingSummary).toBe(true); + }); + + test("omitted reasoning and omitted effort still default to hideThinkingSummary", () => { + const parsed = parseRequest({ + model: "test-model", + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + }); + expect(parsed.options.hideThinkingSummary).toBe(true); + }); + + test("reasoning effort of none defaults to hideThinkingSummary", () => { + const parsed = parseRequest({ + model: "test-model", + reasoning: { effort: "none" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }], + }); + expect(parsed.options.hideThinkingSummary).toBe(true); + }); + + test("combo injected effort defaults summary to auto", () => { + const target: Pick = { + provider: "test-provider", + model: "test-model", + }; + const body = { model: "combo/test", input: [] }; + const child = concreteComboRequestBody(body, target, "high", ["high"]); + expect(child.reasoning).toEqual({ effort: "high", summary: "auto" }); + }); +}); From f6aabb4eb18a43a4008c4616ee2b34ec45a1b7d3 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Wed, 23 Sep 2026 06:23:44 +0900 Subject: [PATCH 05/24] fix: bound Fernet slot runs, Kiro error-body read, and skill-path line slice (#5310) Carries #5310 onto current dev. The follow-up commit makes the Fernet run cap fail closed and moves the Kiro regression out of the capped stream suite. --- src/adapters/kiro/adapter.ts | 1 + src/adapters/kiro/stream.ts | 4 ++- src/claude/inbound.ts | 11 +++++-- src/server/responses/encrypted-payload.ts | 4 +++ .../claude-integration/claude-inbound.test.ts | 6 ++++ .../multi-agent-compat.test.ts | 15 +++++++++ tests/providers/kiro/kiro-stream.test.ts | 32 +++++++++++++++++++ 7 files changed, 69 insertions(+), 4 deletions(-) diff --git a/src/adapters/kiro/adapter.ts b/src/adapters/kiro/adapter.ts index b6a374cf4cc..0300c52907f 100644 --- a/src/adapters/kiro/adapter.ts +++ b/src/adapters/kiro/adapter.ts @@ -250,6 +250,7 @@ export function createKiroAdapter(provider: OcxProviderConfig): ProviderAdapter }); return { response, + abortSignal: requestAbortSignal, inputTokens: retry.inputTokens, contextInputEstimate: retry.contextInputEstimate, nameMap: retry.nameMap, diff --git a/src/adapters/kiro/stream.ts b/src/adapters/kiro/stream.ts index 0536038e7dd..1955c2785df 100644 --- a/src/adapters/kiro/stream.ts +++ b/src/adapters/kiro/stream.ts @@ -19,6 +19,7 @@ import { noteKiroTransientThrottle } from "../kiro-retry"; import { InlineThinkTagParser } from "../inline-think-tags"; import { isCompleteKiroToolInput, kiroTruncationErrorMessage } from "../kiro-truncation"; import { isValidKiroConversationId } from "../kiro-wire"; +import { readDisplaySafeErrorPayloadText } from "../upstream-http-error"; import { tagKiroReasoningBlob } from "./reasoning"; import { estimateKiroTokens, kiroUpstreamContextWindow } from "./usage"; @@ -74,6 +75,7 @@ function createKiroAttemptRetention(budget: TranslatorBudget): KiroAttemptRetent interface KiroFallbackAttempt { response: Response; + abortSignal?: AbortSignal; inputTokens: number; contextInputEstimate: number; nameMap: Map; @@ -1092,7 +1094,7 @@ export async function* parseKiroStream( firstResult.releaseRetained(); fallback.releaseRequestBody?.(); if (!fallback.response.ok) { - const payload = await fallback.response.text().catch(() => ""); + const payload = await readDisplaySafeErrorPayloadText(fallback.response, fallback.abortSignal); const failure = classifyKiroHttpError(fallback.response.status, fallback.response.headers, payload); yield { type: "error", diff --git a/src/claude/inbound.ts b/src/claude/inbound.ts index 2938c9b2a9d..489075b1e5b 100644 --- a/src/claude/inbound.ts +++ b/src/claude/inbound.ts @@ -118,6 +118,7 @@ export function effectiveBlockedSkillNames(cc?: Pick SKILL_TEXT_PATH_MAX_CHARS) return text; + const dir = pathPrefix.slice(0, firstLineEnd === -1 ? pathPrefix.length : firstLineEnd).trim(); // Windows clients send `C:\Users\...\claude-api`; normalize separators before // basenaming (repo precedent: src/codex/inject.ts isOpencodexCatalogPath). - const base = dir.replace(/\\/g, "/").split("/").filter(Boolean).pop()?.toLowerCase() ?? ""; + const normalizedDir = dir.replace(/\\/g, "/").replace(/\/+$/, ""); + const base = normalizedDir.slice(normalizedDir.lastIndexOf("/") + 1).toLowerCase(); if (!names.includes(base)) return text; return `[opencodex] '${base}' skill document bundle (${text.length} chars) elided for routed models ` + "(claudeCode.blockedSkills). The skill is loaded; answer from general knowledge instead of citing the bundle."; diff --git a/src/server/responses/encrypted-payload.ts b/src/server/responses/encrypted-payload.ts index 9835e851cca..9e5e71ee46b 100644 --- a/src/server/responses/encrypted-payload.ts +++ b/src/server/responses/encrypted-payload.ts @@ -123,6 +123,9 @@ function looksLikeUnknownOpaqueSlot(payload: string): boolean { */ const FERNET_TOKEN_CANDIDATE = /g[A-Za-z0-9_-]{97,}={0,2}/g; const FERNET_TOKEN_BOUNDARY_CHAR = /[A-Za-z0-9_=-]/; +// A normal mixed agent task contains one encrypted body. Keep pathological slots +// from amplifying into an attacker-controlled number of request parts. +const MAX_EMBEDDED_FERNET_RUNS_PER_SLOT = 64; interface FernetTokenRun { index: number; @@ -173,6 +176,7 @@ function fernetTokenRuns(payload: string): FernetTokenRun[] { if (after && FERNET_TOKEN_BOUNDARY_CHAR.test(after)) continue; if (!isStructurallyValidFernetToken(token)) continue; runs.push({ index, token }); + if (runs.length >= MAX_EMBEDDED_FERNET_RUNS_PER_SLOT) break; } return runs; } diff --git a/tests/claude-integration/claude-inbound.test.ts b/tests/claude-integration/claude-inbound.test.ts index 5631ec1271a..308f06b0c80 100644 --- a/tests/claude-integration/claude-inbound.test.ts +++ b/tests/claude-integration/claude-inbound.test.ts @@ -675,6 +675,12 @@ describe("bundled-skill elision for routed models (devlog 260712 060)", () => { const texts = userTexts(requestWithSkillTextBlock("claude-api", 500_000, undefined, "C:claude-api")); expect(texts.some(t => t.length > 400_000)).toBe(true); }); + + test("text-block carrier: oversized marker paths pass through without unbounded parsing", () => { + const oversizedDir = `/${"/".repeat(10_000)}claude-api`; + const texts = userTexts(requestWithSkillTextBlock("claude-api", 20_000, undefined, oversizedDir)); + expect(texts.some(t => t.startsWith(`Base directory for this skill: ${oversizedDir}`))).toBe(true); + }); }); describe("ocx-route directive (devlog 072)", () => { diff --git a/tests/codex-integration/multi-agent-compat.test.ts b/tests/codex-integration/multi-agent-compat.test.ts index d4c96d8513e..1519c53a05b 100644 --- a/tests/codex-integration/multi-agent-compat.test.ts +++ b/tests/codex-integration/multi-agent-compat.test.ts @@ -1478,6 +1478,21 @@ describe("sanitizeEncryptedContentInPlace", () => { const parts = (input[0] as { content: Array> }).content; expect(parts[0]).toEqual({ type: "encrypted_content", encrypted_content: fernet }); }); + + test("mixed slots cap Fernet expansion", () => { + const payload = Array.from({ length: 1_000 }, () => fernetFixture()).join("."); + const input = [ + { type: "message", role: "user", content: [ + { type: "encrypted_content", encrypted_content: `preamble.${payload}` }, + ] }, + ]; + + expect(sanitizeEncryptedContentInPlace(input)).toBe(1); + const parts = (input[0] as { content: Array> }).content; + expect(parts.length).toBeLessThanOrEqual(129); + expect(parts.filter(part => part.type === "encrypted_content")).toHaveLength(64); + expect(parts.at(-1)?.type).toBe("input_text"); + }); }); describe("spawn-message delivery (agent_message + encrypted slot)", () => { diff --git a/tests/providers/kiro/kiro-stream.test.ts b/tests/providers/kiro/kiro-stream.test.ts index 5ad687ef6f7..653575ccd4a 100644 --- a/tests/providers/kiro/kiro-stream.test.ts +++ b/tests/providers/kiro/kiro-stream.test.ts @@ -369,6 +369,38 @@ describe("kiro adapter — parseStream", () => { }); }); + test("fallback HTTP errors stop reading oversized upstream bodies", async () => { + const chunk = new TextEncoder().encode("A".repeat(32 * 1024)); + let pulls = 0; + let cancelled = false; + globalThis.fetch = (async () => new Response(new ReadableStream({ + pull(controller) { + pulls += 1; + controller.enqueue(chunk); + }, + cancel() { + cancelled = true; + }, + }), { + status: 400, + headers: { "content-type": "text/plain" }, + })) as typeof fetch; + const adapter = createKiroAdapter(provider); + await adapter.buildRequest(parsedWith([{ role: "user", content: "do it" }], [bashTool])); + + const events = await collectAdapterEvents(adapter.parseStream(new Response(streamOf( + eventFrame({ content: "I am checking." }), + )))); + + expect(cancelled).toBe(true); + expect(pulls).toBeLessThan(10); + expect(events.at(-1)).toMatchObject({ + type: "error", + status: 400, + retryable: false, + }); + }); + test("large first-attempt text stays charged through fallback construction and releases after parse", async () => { const budget = createTranslatorBudget(); const firstText = "x".repeat(10 * 1024 * 1024); From 5a4f8a057f3705d4d43fbcdebefb6810d546d85b Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 06:27:53 +0900 Subject: [PATCH 06/24] docs(reasoning): reconcile inline-tag whitespace contract Interleaved inline-tag parsing preserves answer whitespace; only Kiro single-block mode drops the whitespace after its leading block. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- structure/providers/chat-compat.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index e71e3a67681..dad2bc581c7 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -64,9 +64,9 @@ inside subsequent code fences or quoted examples: this explicit opt-in does not Gateways producing ambiguous literals should use structured reasoning instead. Iterative draining keeps stack depth independent of the number of blocks in an upstream chunk. A block left unterminated at end of stream flushes as reasoning rather than being -dropped. Whitespace between the leading block and the start of the answer is dropped as formatting -noise; once answer text has been emitted, whitespace after a later closing tag is preserved, -because a mid-answer block sits inside markdown or code where indentation is meaningful. +dropped. In interleaved Chat mode, whitespace after the leading block and after later blocks +remains answer text, including indentation and blank lines. Kiro uses single-block mode and +retains its existing first-answer normalization. `src/adapters/inline-think-tags.ts` owns the parser and is shared with the Kiro adapter, which consumes it in single-block mode with its existing leading/first-answer normalization. Regression coverage is in `tests/adapters/openai/openai-chat-inline-think-tags.test.ts` and From 77f97ffa61a9ac6990a328d35a4bf66b815add27 Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 06:27:53 +0900 Subject: [PATCH 07/24] fix(responses): fail closed on Fernet run overflow and keep the Kiro suite under its cap A slot with more than 64 structurally valid Fernet runs is now treated as unreadable or omitted as a whole, so no unexamined tail reaches the provider as text. The bounded Kiro fallback error-body regression moves byte for byte into a registered sibling file, and the Kiro, Responses and inbound contracts document the new bounds. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- scripts/test-layout/layout.json | 1 + src/server/responses/encrypted-payload.ts | 30 +++-- structure/data-planes/inbound-compat.md | 7 ++ structure/providers/kiro.md | 8 ++ structure/runtime.md | 1 + structure/transports/responses.md | 7 ++ .../multi-agent-compat.test.ts | 67 +++++++++-- tests/fixtures/test-layout-expected.json | 1 + .../kiro/kiro-fallback-error-body.test.ts | 113 ++++++++++++++++++ tests/providers/kiro/kiro-stream.test.ts | 32 ----- .../server/v2-agent-message-failfast.test.ts | 26 +++- 11 files changed, 238 insertions(+), 55 deletions(-) create mode 100644 tests/providers/kiro/kiro-fallback-error-body.test.ts diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 3513627da15..a9728a64cf2 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -886,6 +886,7 @@ "kiro-auth-context-continuation.test.ts": "providers/kiro", "kiro-builder-id-profile.test.ts": "providers/kiro", "kiro-calibration.test.ts": "providers/kiro", + "kiro-fallback-error-body.test.ts": "providers/kiro", "kiro-images.test.ts": "providers/kiro", "kiro-oauth.test.ts": "providers/kiro", "kiro-pool-rank.test.ts": "providers/kiro", diff --git a/src/server/responses/encrypted-payload.ts b/src/server/responses/encrypted-payload.ts index 9e5e71ee46b..82da49bca95 100644 --- a/src/server/responses/encrypted-payload.ts +++ b/src/server/responses/encrypted-payload.ts @@ -131,6 +131,10 @@ interface FernetTokenRun { index: number; token: string; } +interface FernetRunScan { + runs: FernetTokenRun[]; + overflow: boolean; +} /** * Validate only the key-independent Fernet wire structure. Authenticity cannot be @@ -161,11 +165,11 @@ function isStructurallyValidFernetToken(token: string): boolean { } export function structurallyValidFernetTokens(payload: string): string[] { - return fernetTokenRuns(payload).map(run => run.token); + return fernetTokenRuns(payload).runs.map(run => run.token); } /** Maximal, boundary-delimited and structurally valid Fernet runs embedded in a slot. */ -function fernetTokenRuns(payload: string): FernetTokenRun[] { +function fernetTokenRuns(payload: string): FernetRunScan { const runs: FernetTokenRun[] = []; for (const match of payload.matchAll(FERNET_TOKEN_CANDIDATE)) { const index = match.index ?? 0; @@ -175,10 +179,10 @@ function fernetTokenRuns(payload: string): FernetTokenRun[] { if (before && FERNET_TOKEN_BOUNDARY_CHAR.test(before)) continue; if (after && FERNET_TOKEN_BOUNDARY_CHAR.test(after)) continue; if (!isStructurallyValidFernetToken(token)) continue; + if (runs.length === MAX_EMBEDDED_FERNET_RUNS_PER_SLOT) return { runs, overflow: true }; runs.push({ index, token }); - if (runs.length >= MAX_EMBEDDED_FERNET_RUNS_PER_SLOT) break; } - return runs; + return { runs, overflow: false }; } function textWithoutFernetRuns(payload: string, runs: readonly FernetTokenRun[]): string { @@ -295,7 +299,9 @@ export function hasUnreadableEncryptedAgentTask(input: unknown): boolean { } if (fragmentParts.has(part)) continue; - const runs = fernetTokenRuns(record.encrypted_content); + const scan = fernetTokenRuns(record.encrypted_content); + if (scan.overflow) return true; + const { runs } = scan; if (runs.length > 0) hasFernetTask = true; readableParts.push(textWithoutFernetRuns(record.encrypted_content, runs)); } @@ -312,9 +318,11 @@ export function hasUnreadableEncryptedAgentTask(input: unknown): boolean { export function encryptedSlotParts(payload: string): Array> { + const scan = fernetTokenRuns(payload); + if (scan.overflow) return [{ type: "input_text", text: OMITTED_ENCRYPTED_CONTENT_TEXT }]; const parts: Array> = []; let last = 0; - for (const run of fernetTokenRuns(payload)) { + for (const run of scan.runs) { const before = payload.slice(last, run.index); if (before.trim().length > 0) parts.push({ type: "input_text", text: before }); parts.push({ type: "encrypted_content", encrypted_content: run.token }); @@ -424,7 +432,9 @@ export function stripAgentMessageCiphertextInPlace(input: unknown): number { /** Free text: drop embedded token runs, and replace a slot that is nothing but a token. */ function textWithoutCiphertext(text: string): string { - const runs = fernetTokenRuns(text); + const scan = fernetTokenRuns(text); + if (scan.overflow) return OMITTED_ENCRYPTED_CONTENT_TEXT; + const { runs } = scan; if (runs.length > 0) return textWithRunsOmitted(text, runs); return looksLikeFernetToken(text.trim()) ? OMITTED_ENCRYPTED_CONTENT_TEXT : text; } @@ -477,11 +487,11 @@ function contentWithoutCiphertext(content: unknown[]): unknown[] { changed = true; // Keep whatever plaintext a recognizable slot carries around its token; a slot this // cannot parse is replaced whole rather than forwarded on the chance that it is benign. - const runs = fernetTokenRuns(record.encrypted_content); + const scan = fernetTokenRuns(record.encrypted_content); return { type: "input_text", - text: runs.length > 0 - ? textWithRunsOmitted(record.encrypted_content, runs) + text: scan.overflow ? OMITTED_ENCRYPTED_CONTENT_TEXT : scan.runs.length > 0 + ? textWithRunsOmitted(record.encrypted_content, scan.runs) : OMITTED_ENCRYPTED_CONTENT_TEXT, }; } diff --git a/structure/data-planes/inbound-compat.md b/structure/data-planes/inbound-compat.md index a1cad67deae..3cb935e63bb 100644 --- a/structure/data-planes/inbound-compat.md +++ b/structure/data-planes/inbound-compat.md @@ -275,6 +275,13 @@ Instruction notice extraction scans fence ranges once and walks original lines b a decreasing cursor. It accepts exactly one ASCII space inside the token notice, preserves unmatched prefix bytes, and does not repeatedly scan or copy shrinking prompt prefixes. +## Claude skill marker path bound + +`src/claude/inbound.ts` examines at most 4,097 characters after the skill base-directory +marker when deciding whether to elide a large blocked skill bundle. A marker path longer +than 4,096 characters passes through unchanged; a recognized bounded path retains the +existing elision behavior. This bound applies to translated Claude Messages ingress. + Native Chat applies qualifying effort ceilings independently of model pins; pin selection precedes the cap and only pins or cap rewrites enter wire mapping. The [catalog effort contract](../catalog.md#ultra-reasoning-level) records the V1/compaction exemptions and caller-preservation boundary. Pool quota producers and account commands follow the [bounded raw-observation contract](../providers/openai-tiers.md#bounded-pool-quota-observations), separate from the latest display snapshot and capacity estimates. diff --git a/structure/providers/kiro.md b/structure/providers/kiro.md index b1d0b503d55..f1664168159 100644 --- a/structure/providers/kiro.md +++ b/structure/providers/kiro.md @@ -37,6 +37,14 @@ raw body. > Decision record: [ADR-0061](../decisions/ADR-0061-kiro-responses-text-controls.md) +## Bounded fallback HTTP errors + +When a first Kiro stream needs a completion fallback, the fallback response's non-success +body is read through the shared display-safe bounded reader with the attempt's abort signal. +The adapter emits an error with the upstream status and does not emit a successful completion. +A body that exceeds the reader's limit is cancelled and cannot contribute unbounded text to +the error message. Coverage: `tests/providers/kiro/kiro-fallback-error-body.test.ts`. + ## Kiro reasoning round-trip (`signature`) Kiro never returns plaintext reasoning for its **GPT-5.6 family** (`gpt-5.6-sol`, `-terra`, diff --git a/structure/runtime.md b/structure/runtime.md index 1b2d8eaf338..fcf5c510078 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -446,6 +446,7 @@ following a final symlink, so an exchange during a mutation cannot redirect the `claudeCode.stabilizePromptCache` is a default-off operator setting for [translated instruction stabilization](data-planes/inbound-compat.md#opt-in-claude-instruction-stabilization). Config JSON preserves the boolean; only literal true activates the role-changing transform. +Claude skill-bundle marker parsing follows the [bounded inbound contract](data-planes/inbound-compat.md#claude-skill-marker-path-bound). The lightweight top-level CLI help counts Cline CLI among the fifteen registered export clients; registry parity remains covered by the client help and integration tests. Devin CLI credential path composition in `src/oauth/devin/cli-import.ts` follows the selected platform: Windows uses Win32 APPDATA paths, other platforms use POSIX XDG-data paths. The explicit absolute override remains verbatim; credential parsing and login behavior are unchanged. The `src/providers/devin-provider-merge-migration.ts` startup migration treats the legacy provider row and its OAuth slot as one account-bound unit: an occupied destination or a refused config projection leaves both unchanged, and both backups complete before either file changes. The adapter takes a tenant host only from the stored account that owns the exact key being transmitted, in the literal slot or, during a detached rekey window, the alias slot, so separately configured or forwarded credentials and non-owning accounts cannot lend another account's destination. diff --git a/structure/transports/responses.md b/structure/transports/responses.md index 20d5871f9b3..41499dd4632 100644 --- a/structure/transports/responses.md +++ b/structure/transports/responses.md @@ -353,6 +353,13 @@ decrypt/decode identity enters one request-budgeted sanitize-and-rebuild attempt pre-commit SSE/WebSocket terminal envelopes; the single-shot guard remains armed on the rebuilt send. +A mixed `encrypted_content` slot may contain structurally valid Fernet runs alongside text. +`src/server/responses/encrypted-payload.ts` recognizes at most 64 runs per slot. Finding a +65th marks the slot as overflow: sanitization and agent-message stripping replace that +whole slot with `[encrypted content omitted]`, while unreadable-task detection remains +fail-closed. The scanner never emits an unexamined suffix as text. The limit constrains +part expansion without changing single-token replay or the separate 32-part task-recovery cap. + > Decision record: [ADR-5236](../decisions/ADR-5236-responses-http-sse.md) Codex pool account changes are a separate portability question from destination serving identity. diff --git a/tests/codex-integration/multi-agent-compat.test.ts b/tests/codex-integration/multi-agent-compat.test.ts index 1519c53a05b..dc7cfbb9e73 100644 --- a/tests/codex-integration/multi-agent-compat.test.ts +++ b/tests/codex-integration/multi-agent-compat.test.ts @@ -7,7 +7,7 @@ import { afterAll, afterEach, beforeAll, describe, expect, test } from "bun:test import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; -import { injectDeveloperMessage, multiAgentGuidanceText, sanitizeEncryptedContentInPlace } from "../../src/server/responses"; +import { injectDeveloperMessage, multiAgentGuidanceText, sanitizeEncryptedContentInPlace, stripAgentMessageCiphertextInPlace } from "../../src/server/responses"; import { MULTI_AGENT_MODE_HINT_RECOMMENDATION } from "../../src/codex/multi-agent-mode-policy"; import { parseRequest } from "../../src/responses/parser"; import type { OcxParsedRequest } from "../../src/types"; @@ -1368,10 +1368,11 @@ describe("injectDeveloperMessage", () => { }); describe("sanitizeEncryptedContentInPlace", () => { - const fernetFixture = (): string => { + const fernetFixture = (index = 0): string => { const raw = Buffer.alloc(73, 0x5a); raw[0] = 0x80; raw.writeBigUInt64BE(1_720_000_000n, 1); + raw.writeUInt32BE(index, 25); const unpadded = raw.toString("base64url"); return `${unpadded}${"=".repeat((4 - (unpadded.length % 4)) % 4)}`; }; @@ -1479,19 +1480,61 @@ describe("sanitizeEncryptedContentInPlace", () => { expect(parts[0]).toEqual({ type: "encrypted_content", encrypted_content: fernet }); }); - test("mixed slots cap Fernet expansion", () => { - const payload = Array.from({ length: 1_000 }, () => fernetFixture()).join("."); - const input = [ - { type: "message", role: "user", content: [ - { type: "encrypted_content", encrypted_content: `preamble.${payload}` }, - ] }, - ]; + test("65 valid runs omit the mixed slot without emitting a token suffix", () => { + const tokens = Array.from({ length: 65 }, (_, index) => fernetFixture(index)); + const withinLimit = [{ type: "message", role: "user", content: [ + { type: "encrypted_content", encrypted_content: `preamble.${tokens.slice(0, 64).join(".")}` }, + ] }]; + expect(sanitizeEncryptedContentInPlace(withinLimit)).toBe(1); + const withinParts = (withinLimit[0] as { content: Array> }).content; + expect(withinParts.length).toBeLessThanOrEqual(129); + expect(withinParts.filter(part => part.type === "encrypted_content")).toHaveLength(64); + const input = [{ type: "message", role: "user", content: [ + { type: "encrypted_content", encrypted_content: `preamble.${tokens.join(".")}` }, + ] }]; expect(sanitizeEncryptedContentInPlace(input)).toBe(1); const parts = (input[0] as { content: Array> }).content; - expect(parts.length).toBeLessThanOrEqual(129); - expect(parts.filter(part => part.type === "encrypted_content")).toHaveLength(64); - expect(parts.at(-1)?.type).toBe("input_text"); + expect(parts).toEqual([{ type: "input_text", text: "[encrypted content omitted]" }]); + const serialized = JSON.stringify(input); + for (const token of tokens) expect(serialized).not.toContain(token); + }); + + test("65 valid runs in an agent message normalize without forwarding token text", () => { + const tokens = Array.from({ length: 65 }, (_, index) => fernetFixture(index)); + const input = [{ type: "agent_message", id: "a1", author: "/root", recipient: "/root/worker", content: [ + { type: "encrypted_content", encrypted_content: `preamble.${tokens.join(".")}` }, + ] }]; + + expect(sanitizeEncryptedContentInPlace(input)).toBe(1); + expect(input[0]).toMatchObject({ type: "message", role: "user" }); + expect(stripAgentMessageCiphertextInPlace(input)).toBe(0); + const serialized = JSON.stringify(input); + for (const token of tokens) expect(serialized).not.toContain(token); + }); + + test("65 valid runs in a raw agent slot are omitted before lowering", () => { + const tokens = Array.from({ length: 65 }, (_, index) => fernetFixture(index)); + const input = [{ type: "agent_message", content: [ + { type: "encrypted_content", encrypted_content: `preamble.${tokens.join(".")}` }, + ] }]; + expect(stripAgentMessageCiphertextInPlace(input)).toBe(1); + expect((input[0] as { content: unknown[] }).content).toEqual([ + { type: "input_text", text: "[encrypted content omitted]" }, + ]); + for (const token of tokens) expect(JSON.stringify(input)).not.toContain(token); + }); + + test("65 valid runs in agent text are omitted before lowering", () => { + const tokens = Array.from({ length: 65 }, (_, index) => fernetFixture(index)); + const input = [{ type: "agent_message", content: [ + { type: "input_text", text: `preamble.${tokens.join(".")}` }, + ] }]; + expect(stripAgentMessageCiphertextInPlace(input)).toBe(1); + expect((input[0] as { content: unknown[] }).content).toEqual([ + { type: "input_text", text: "[encrypted content omitted]" }, + ]); + for (const token of tokens) expect(JSON.stringify(input)).not.toContain(token); }); }); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 089074d49b2..9af6722090f 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -713,6 +713,7 @@ "kiro-auth-context-continuation.test.ts": "providers/kiro", "kiro-builder-id-profile.test.ts": "providers/kiro", "kiro-calibration.test.ts": "providers/kiro", + "kiro-fallback-error-body.test.ts": "providers/kiro", "kiro-images.test.ts": "providers/kiro", "kiro-oauth.test.ts": "providers/kiro", "kiro-pool-rank.test.ts": "providers/kiro", diff --git a/tests/providers/kiro/kiro-fallback-error-body.test.ts b/tests/providers/kiro/kiro-fallback-error-body.test.ts new file mode 100644 index 00000000000..588e0f3f2b3 --- /dev/null +++ b/tests/providers/kiro/kiro-fallback-error-body.test.ts @@ -0,0 +1,113 @@ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { mkdtempSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createKiroAdapter as createKiroAdapterProduction } from "../../../src/adapters/kiro"; +import { resetKiroThrottleStateForTests } from "../../../src/adapters/kiro-retry"; +import { clearDebugSetting, getDebugSettings, setDebugSettings } from "../../../src/lib/debug-settings"; +import { encodeMessage } from "../../../src/lib/eventstream-decoder"; +import type { AdapterEvent, OcxParsedRequest, OcxProviderConfig } from "../../../src/types"; +import { removeTreeWithRetry } from "../../helpers/remove-tree"; +import { withTestTranslatorBudget } from "../../helpers/translator-budget"; + +const realFetch = globalThis.fetch; +const originalHome = process.env.HOME; +const originalRegion = process.env.KIRO_REGION; +const originalApiRegion = process.env.KIRO_API_REGION; +const originalArn = process.env.KIRO_PROFILE_ARN; +const originalCredsFile = process.env.KIRO_CREDS_FILE; +const originalCredentialsFile = process.env.KIRO_CREDENTIALS_FILE; +let originalDebug: string | undefined; +let originalDebugFrames: string | undefined; +let originalDebugOverride: boolean | undefined; +let tempHome: string; +beforeEach(() => { + originalDebug = process.env.OCX_DEBUG; + originalDebugFrames = process.env.OCX_DEBUG_FRAMES; + originalDebugOverride = getDebugSettings().runtimeOverride.debug; + tempHome = mkdtempSync(join(tmpdir(), "kiro-fallback-error-")); + process.env.HOME = tempHome; + process.env.KIRO_REGION = "us-east-1"; + delete process.env.KIRO_API_REGION; + delete process.env.KIRO_PROFILE_ARN; + delete process.env.KIRO_CREDS_FILE; + delete process.env.KIRO_CREDENTIALS_FILE; + delete process.env.OCX_DEBUG; + delete process.env.OCX_DEBUG_FRAMES; + clearDebugSetting("debug"); +}); +afterEach(() => { + globalThis.fetch = realFetch; + resetKiroThrottleStateForTests(); + if (originalHome === undefined) delete process.env.HOME; else process.env.HOME = originalHome; + if (originalRegion === undefined) delete process.env.KIRO_REGION; else process.env.KIRO_REGION = originalRegion; + if (originalApiRegion === undefined) delete process.env.KIRO_API_REGION; else process.env.KIRO_API_REGION = originalApiRegion; + if (originalArn === undefined) delete process.env.KIRO_PROFILE_ARN; else process.env.KIRO_PROFILE_ARN = originalArn; + if (originalCredsFile === undefined) delete process.env.KIRO_CREDS_FILE; else process.env.KIRO_CREDS_FILE = originalCredsFile; + if (originalCredentialsFile === undefined) delete process.env.KIRO_CREDENTIALS_FILE; else process.env.KIRO_CREDENTIALS_FILE = originalCredentialsFile; + if (originalDebug === undefined) delete process.env.OCX_DEBUG; else process.env.OCX_DEBUG = originalDebug; + if (originalDebugFrames === undefined) delete process.env.OCX_DEBUG_FRAMES; else process.env.OCX_DEBUG_FRAMES = originalDebugFrames; + if (originalDebugOverride === undefined) clearDebugSetting("debug"); + else setDebugSettings({ debug: originalDebugOverride }); + removeTreeWithRetry(tempHome); +}); + +const provider = { adapter: "kiro", baseUrl: "https://127.0.0.1", authMode: "oauth", apiKey: "tok-123" } as unknown as OcxProviderConfig; +const bashTool = { name: "bash", description: "Run a shell command", parameters: { type: "object" } }; +function createKiroAdapter(...args: Parameters) { + return withTestTranslatorBudget(createKiroAdapterProduction(...args)); +} +function parsedWith(messages: unknown[], tools?: unknown[]): OcxParsedRequest { + return { modelId: "claude-sonnet-4.5", stream: true, options: {}, context: { messages, tools } } as unknown as OcxParsedRequest; +} +const eventFrame = (obj: unknown) => encodeMessage( + { ":message-type": "event", ":event-type": "assistantResponseEvent" }, + new TextEncoder().encode(JSON.stringify(obj)), +); +function streamOf(...frames: Uint8Array[]): ReadableStream { + let i = 0; + return new ReadableStream({ + pull(controller) { + if (i < frames.length) controller.enqueue(frames[i++]); + else controller.close(); + }, + }); +} +async function collectAdapterEvents(events: AsyncGenerator): Promise { + const out: AdapterEvent[] = []; + for await (const event of events) out.push(event); + return out; +} + +test("fallback HTTP errors stop reading oversized upstream bodies", async () => { + const chunk = new TextEncoder().encode("A".repeat(32 * 1024)); + let pulls = 0; + let cancelled = false; + globalThis.fetch = (async () => new Response(new ReadableStream({ + pull(controller) { + pulls += 1; + controller.enqueue(chunk); + }, + cancel() { + cancelled = true; + }, + }), { + status: 400, + headers: { "content-type": "text/plain" }, + })) as typeof fetch; + const adapter = createKiroAdapter(provider); + await adapter.buildRequest(parsedWith([{ role: "user", content: "do it" }], [bashTool])); + + const events = await collectAdapterEvents(adapter.parseStream(new Response(streamOf( + eventFrame({ content: "I am checking." }), + )))); + + expect(cancelled).toBe(true); + expect(pulls).toBeLessThan(10); + expect(events.at(-1)).toMatchObject({ + type: "error", + status: 400, + retryable: false, + }); + expect(events.some(event => event.type === "done")).toBe(false); +}); diff --git a/tests/providers/kiro/kiro-stream.test.ts b/tests/providers/kiro/kiro-stream.test.ts index 653575ccd4a..5ad687ef6f7 100644 --- a/tests/providers/kiro/kiro-stream.test.ts +++ b/tests/providers/kiro/kiro-stream.test.ts @@ -369,38 +369,6 @@ describe("kiro adapter — parseStream", () => { }); }); - test("fallback HTTP errors stop reading oversized upstream bodies", async () => { - const chunk = new TextEncoder().encode("A".repeat(32 * 1024)); - let pulls = 0; - let cancelled = false; - globalThis.fetch = (async () => new Response(new ReadableStream({ - pull(controller) { - pulls += 1; - controller.enqueue(chunk); - }, - cancel() { - cancelled = true; - }, - }), { - status: 400, - headers: { "content-type": "text/plain" }, - })) as typeof fetch; - const adapter = createKiroAdapter(provider); - await adapter.buildRequest(parsedWith([{ role: "user", content: "do it" }], [bashTool])); - - const events = await collectAdapterEvents(adapter.parseStream(new Response(streamOf( - eventFrame({ content: "I am checking." }), - )))); - - expect(cancelled).toBe(true); - expect(pulls).toBeLessThan(10); - expect(events.at(-1)).toMatchObject({ - type: "error", - status: 400, - retryable: false, - }); - }); - test("large first-attempt text stays charged through fallback construction and releases after parse", async () => { const budget = createTranslatorBudget(); const firstText = "x".repeat(10 * 1024 * 1024); diff --git a/tests/server/v2-agent-message-failfast.test.ts b/tests/server/v2-agent-message-failfast.test.ts index 18aba6211bf..0ff7259c764 100644 --- a/tests/server/v2-agent-message-failfast.test.ts +++ b/tests/server/v2-agent-message-failfast.test.ts @@ -19,10 +19,11 @@ const takeSpendHome = (): void => { releaseSpendHome ??= acquireOwnedSpendHome() * block + HMAC. Bytes are synthetic, so this validates wire shape without * publishing a real captured task or claiming the HMAC is authentic. */ -function fernetFixture(ciphertextBytes = 16, version = 0x80): string { +function fernetFixture(ciphertextBytes = 16, version = 0x80, variant = 0): string { const raw = Buffer.alloc(57 + ciphertextBytes, 0x5a); raw[0] = version; raw.writeBigUInt64BE(1_720_000_000n, 1); + if (variant !== 0) raw.writeUInt32BE(variant, 25); const unpadded = raw.toString("base64url"); return `${unpadded}${"=".repeat((4 - (unpadded.length % 4)) % 4)}`; } @@ -230,6 +231,29 @@ describe("V2 routed agent-message ciphertext guard", () => { expect(raw).not.toContain("gAAAA"); }); + test("65 Fernet runs in a tail task are refused before upstream dispatch", async () => { + const tokens = Array.from({ length: 65 }, (_, index) => fernetFixture(16, 0x80, index)); + const input = agentMessage([ + { type: "input_text", text: ROUTING_ENVELOPE }, + { type: "encrypted_content", encrypted_content: tokens.join(".") }, + ]); + expect(new Set(tokens).size).toBe(65); + expect(hasUnreadableEncryptedAgentTask(input)).toBe(true); + + let fetchCalls = 0; + globalThis.fetch = (async () => { + fetchCalls += 1; + throw new Error("provider dispatch must not happen"); + }) as typeof fetch; + + const response = await post(routedConfig(), "xai/grok-4.5", input); + expect(response.status).toBe(400); + expect(await response.json()).toMatchObject({ + error: { type: "invalid_request_error", code: "unreadable_encrypted_agent_task" }, + }); + expect(fetchCalls).toBe(0); + }); + test("filters a combo to a decrypt-capable native target before dispatch", async () => { const fetchedUrls: string[] = []; const nativeToken = fakeChatGptJwt({ chatgpt_account_id: "native-combo-caller" }); From d2432d043c44db4cf38e92596f648c9cd78350d7 Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 06:40:17 +0900 Subject: [PATCH 08/24] fix(reasoning): scan inline think tags with a moving cursor The parser copied, rescanned and reserved the whole remaining response after every block, so one upstream chunk carrying many short blocks cost quadratic work. It now scans each chunk from an offset and charges the translator budget only for retained carry: undecided leading input or a trailing tag fragment. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- src/adapters/inline-think-tags.ts | 155 ++++++++---------- structure/providers/chat-compat.md | 5 +- .../openai/inline-think-boundaries.test.ts | 24 ++- 3 files changed, 89 insertions(+), 95 deletions(-) diff --git a/src/adapters/inline-think-tags.ts b/src/adapters/inline-think-tags.ts index 2bf7a481740..8c6470f1bd0 100644 --- a/src/adapters/inline-think-tags.ts +++ b/src/adapters/inline-think-tags.ts @@ -64,32 +64,35 @@ export class InlineThinkTagParser { feed(text: string): AdapterEvent[] { if (!text) return []; if (this.state === "streaming") return [{ type: "text_delta", text }]; - if (this.state === "thinking") { - this.replaceCarry("thinkingBuffer", this.thinkingBuffer + text); - return this.drain(); - } - if (this.state === "scanning") { - this.replaceCarry("preBuffer", this.preBuffer + text); - return this.drain(); + let input = text; + if (this.state === "pre") { + input = this.preBuffer + text; + this.replaceCarry("preBuffer", ""); + const stripped = input.trimStart(); + const openTag = OPEN_TAGS.find(tag => stripped.startsWith(tag)); + if (openTag) { + const leadingLength = input.length - stripped.length; + const leading = this.interleaved ? input.slice(0, leadingLength) : ""; + this.state = "thinking"; + this.closeTag = closeTagFor(openTag); + const events: AdapterEvent[] = leading ? [{ type: "text_delta", text: leading }] : []; + return this.drainChunk(input, leadingLength + openTag.length, events); + } + if (stripped.length <= MAX_OPEN_TAG && isPossibleOpenTagPrefix(stripped)) { + this.replaceCarry("preBuffer", input); + return []; + } + this.state = "streaming"; + return input ? [{ type: "text_delta", text: input }] : []; } - this.replaceCarry("preBuffer", this.preBuffer + text); - const stripped = this.preBuffer.trimStart(); - const openTag = OPEN_TAGS.find(tag => stripped.startsWith(tag)); - if (openTag) { - const leading = this.interleaved ? this.preBuffer.slice(0, this.preBuffer.length - stripped.length) : ""; - this.state = "thinking"; - this.closeTag = closeTagFor(openTag); - this.replaceCarry("thinkingBuffer", stripped.slice(openTag.length)); + if (this.state === "thinking") { + input = this.thinkingBuffer + text; + this.replaceCarry("thinkingBuffer", ""); + } else { + input = this.preBuffer + text; this.replaceCarry("preBuffer", ""); - const events: AdapterEvent[] = leading ? [{ type: "text_delta", text: leading }] : []; - for (const event of this.drain()) events.push(event); - return events; } - if (stripped.length <= MAX_OPEN_TAG && isPossibleOpenTagPrefix(stripped)) return []; - this.state = "streaming"; - const out = this.preBuffer; - this.replaceCarry("preBuffer", ""); - return out ? [{ type: "text_delta", text: out }] : []; + return this.drainChunk(input, 0, []); } flush(): AdapterEvent[] { @@ -116,78 +119,52 @@ export class InlineThinkTagParser { this.state = "streaming"; } - private drain(): AdapterEvent[] { - const events: AdapterEvent[] = []; - // State transitions consume a complete tag; incomplete carry ends this feed. - // Do not recurse for each block in one upstream chunk. + private drainChunk(input: string, start: number, events: AdapterEvent[]): AdapterEvent[] { + let offset = start; for (;;) { - const before = this.state; - const next = before === "thinking" ? this.drainThinking() : this.drainScanning(); - for (const event of next) events.push(event); - if (this.state === before || this.state === "streaming") return events; - } - } - - private drainThinking(): AdapterEvent[] { - const close = this.closeTag; - const idx = this.thinkingBuffer.indexOf(close); - if (idx >= 0) { - const thinking = this.thinkingBuffer.slice(0, idx); - const remainder = this.thinkingBuffer.slice(idx + close.length); - // Opt-in Chat answers are byte-preserving; keep Kiro's legacy normalization. - const after = this.interleaved ? remainder : remainder.trimStart(); - this.replaceCarry("thinkingBuffer", ""); - const events: AdapterEvent[] = []; - if (thinking) events.push({ type: "reasoning_raw_delta", text: thinking }); - if (this.interleaved) { - this.state = "scanning"; - this.replaceCarry("preBuffer", after); - } else { - this.state = "streaming"; - if (after) events.push({ type: "text_delta", text: after }); + if (this.state === "thinking") { + const idx = input.indexOf(this.closeTag, offset); + if (idx >= 0) { + if (idx > offset) events.push({ type: "reasoning_raw_delta", text: input.slice(offset, idx) }); + offset = idx + this.closeTag.length; + if (!this.interleaved) { + // Opt-in Chat answers are byte-preserving; keep Kiro's legacy normalization. + const after = input.slice(offset).trimStart(); + this.state = "streaming"; + if (after) events.push({ type: "text_delta", text: after }); + return events; + } + this.state = "scanning"; + continue; + } + // Keep only a possible close tag, and do not split a surrogate pair. + const cut = Math.max(offset, surrogateSafeCut(input, input.length - MAX_CLOSE_TAG)); + if (cut > offset) events.push({ type: "reasoning_raw_delta", text: input.slice(offset, cut) }); + this.replaceCarry("thinkingBuffer", input.slice(cut)); + return events; } - return events; - } - if (this.thinkingBuffer.length <= MAX_CLOSE_TAG) return []; - // Hold back a possible partial close tag, and never split a surrogate pair - // at the send boundary: a lone high surrogate encodes as U+FFFD. - const cut = surrogateSafeCut(this.thinkingBuffer, this.thinkingBuffer.length - MAX_CLOSE_TAG); - const send = this.thinkingBuffer.slice(0, cut); - this.replaceCarry("thinkingBuffer", this.thinkingBuffer.slice(cut)); - return send ? [{ type: "reasoning_raw_delta", text: send }] : []; - } - /** - * Interleaved mode only: the response already proved it carries inline thinking, so a later - * block can open anywhere in the answer text rather than only at the start. - */ - private drainScanning(): AdapterEvent[] { - const events: AdapterEvent[] = []; - let openIndex = -1; - let openTag: ThinkingTag | undefined; - for (const tag of OPEN_TAGS) { - const index = this.preBuffer.indexOf(tag); - if (index >= 0 && (openIndex < 0 || index < openIndex)) { - openIndex = index; - openTag = tag; + // Interleaved mode has already seen a leading tag; later tags delimit anywhere. + let openIndex = input.indexOf("<", offset); + let openTag: ThinkingTag | undefined; + while (openIndex >= 0) { + openTag = OPEN_TAGS.find(tag => input.startsWith(tag, openIndex)); + if (openTag) break; + openIndex = input.indexOf("<", openIndex + 1); } - } - if (openIndex >= 0 && openTag) { - const before = this.preBuffer.slice(0, openIndex); - if (before) events.push({ type: "text_delta", text: before }); - this.state = "thinking"; - this.closeTag = closeTagFor(openTag); - this.replaceCarry("thinkingBuffer", this.preBuffer.slice(openIndex + openTag.length)); - this.replaceCarry("preBuffer", ""); + if (openIndex >= 0 && openTag) { + if (openIndex > offset) events.push({ type: "text_delta", text: input.slice(offset, openIndex) }); + offset = openIndex + openTag.length; + this.state = "thinking"; + this.closeTag = closeTagFor(openTag); + continue; + } + // Hold back only a possible open-tag prefix, with a surrogate-safe boundary. + const cut = Math.max(offset, surrogateSafeCut(input, input.length - (MAX_OPEN_TAG - 1))); + if (cut > offset) events.push({ type: "text_delta", text: input.slice(offset, cut) }); + this.replaceCarry("preBuffer", input.slice(cut)); return events; } - // Hold back only as much as a partial open tag could occupy. - const cut = surrogateSafeCut(this.preBuffer, this.preBuffer.length - (MAX_OPEN_TAG - 1)); - if (cut > 0) { - events.push({ type: "text_delta", text: this.preBuffer.slice(0, cut) }); - this.replaceCarry("preBuffer", this.preBuffer.slice(cut)); - } - return events; } } diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index dad2bc581c7..8ad48c0d4f1 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -61,8 +61,9 @@ and after every closing tag remains answer text; after it engages it keeps splitting later blocks, because M-series models interleave thinking with answer segments, including same-line interleaving. Once engaged, tags are protocol delimiters even inside subsequent code fences or quoted examples: this explicit opt-in does not parse Markdown. -Gateways producing ambiguous literals should use structured reasoning instead. Iterative draining -keeps stack depth independent of the number of blocks in an upstream chunk. +Gateways producing ambiguous literals should use structured reasoning instead. A moving cursor +scans each upstream chunk without copying the remaining response after every block. Only undecided +leading input or a trailing tag fragment is retained and charged to the translator budget. A block left unterminated at end of stream flushes as reasoning rather than being dropped. In interleaved Chat mode, whitespace after the leading block and after later blocks remains answer text, including indentation and blank lines. Kiro uses single-block mode and diff --git a/tests/adapters/openai/inline-think-boundaries.test.ts b/tests/adapters/openai/inline-think-boundaries.test.ts index 4bda09221d0..c6411e13ad9 100644 --- a/tests/adapters/openai/inline-think-boundaries.test.ts +++ b/tests/adapters/openai/inline-think-boundaries.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, test } from "bun:test"; +import { describe, expect, spyOn, test } from "bun:test"; import { InlineThinkTagParser, splitInlineThinkContent } from "../../../src/adapters/inline-think-tags"; import type { AdapterEvent } from "../../../src/types"; import { createTestTranslatorBudget } from "../../helpers/translator-budget"; @@ -46,6 +46,22 @@ describe("inline thinking format boundaries", () => { expect(split([block.repeat(count)])).toEqual(expected); expect(split(Array(count).fill(block))).toEqual(expected); }); + test("single-chunk blocks reserve carry in proportion to input size", () => { + const count = 12000; + const input = "ra".repeat(count); + const budget = createTestTranslatorBudget({ maxTurnBytes: 1_000_000 }); + const reserve = spyOn(budget, "reserveTransient"); + const parser = new InlineThinkTagParser(budget, { interleaved: true }); + try { + const events = [...parser.feed(input), ...parser.flush()]; + expect(projection(events)).toEqual({ answer: "a".repeat(count), reasoning: "r".repeat(count) }); + const reservedBytes = reserve.mock.calls.reduce((total, [bytes]) => total + bytes, 0); + expect(reservedBytes).toBeLessThanOrEqual(4 * Buffer.byteLength(input)); + } finally { + parser.dispose(); + reserve.mockRestore(); + } + }); test("partial tags and unterminated reasoning flush without loss", () => { expect(split(["why😀 { expect(projection(splitInlineThinkContent(["other"], "model", undefined, content))) .toEqual({ answer: content, reasoning: "" }); }); - test("dispose releases partial carry and an overflow does not relax the budget", () => { + test("dispose releases pending prefix after a retained-carry overflow", () => { const budget = createTestTranslatorBudget({ maxTurnBytes: 128 }); const parser = new InlineThinkTagParser(budget, { interleaved: true }); - parser.feed("partial"); - expect(() => parser.feed("x".repeat(129))).toThrow(); + parser.feed(" ".repeat(100)); + expect(() => parser.feed(" ".repeat(29))).toThrow(); parser.dispose(); expect(budget.snapshot().currentBytes).toBe(0); }); From 8ba8e1eb9d6b1bbd44ea26b6bb4d9e9064dab290 Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 06:45:37 +0900 Subject: [PATCH 09/24] fix(reasoning): keep undecided leading whitespace incremental Before the format was decided, every content delta rebuilt, trimmed and re-reserved the whole leading prefix, so a stream of one-character whitespace deltas cost quadratic work. Leading whitespace is now kept in segments whose bytes are reserved once and joined only when the format is decided or the stream flushes. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- src/adapters/inline-think-tags.ts | 52 +++++++++++++++---- structure/providers/chat-compat.md | 2 + .../openai/inline-think-boundaries.test.ts | 21 ++++++++ 3 files changed, 64 insertions(+), 11 deletions(-) diff --git a/src/adapters/inline-think-tags.ts b/src/adapters/inline-think-tags.ts index 8c6470f1bd0..229bb06c65e 100644 --- a/src/adapters/inline-think-tags.ts +++ b/src/adapters/inline-think-tags.ts @@ -40,6 +40,9 @@ export interface InlineThinkTagOptions { */ export class InlineThinkTagParser { private state: ParserState = "pre"; + private preWhitespaceChunks: string[] = []; + private preWhitespaceLength = 0; + private preWhitespaceBytes = 0; private preBuffer = ""; private thinkingBuffer = ""; private closeTag = ""; @@ -61,29 +64,55 @@ export class InlineThinkTagParser { this.budget?.releaseRetained(previousBytes, { kind: "reasoning" }); } + private appendPreWhitespace(text: string): void { + if (!text) return; + const bytes = Buffer.byteLength(text); + const reservation = this.budget?.reserveTransient(bytes, { kind: "reasoning" }); + this.preWhitespaceChunks.push(text); + this.preWhitespaceLength += text.length; + this.preWhitespaceBytes += bytes; + reservation?.commitRetained(); + } + + private finishPreWhitespace(emit: boolean): string { + if (this.preWhitespaceLength === 0) return ""; + const out = emit ? this.preWhitespaceChunks.join("") : ""; + this.preWhitespaceChunks.length = 0; + this.preWhitespaceLength = 0; + this.budget?.releaseRetained(this.preWhitespaceBytes, { kind: "reasoning" }); + this.preWhitespaceBytes = 0; + return out; + } + feed(text: string): AdapterEvent[] { if (!text) return []; if (this.state === "streaming") return [{ type: "text_delta", text }]; let input = text; if (this.state === "pre") { - input = this.preBuffer + text; - this.replaceCarry("preBuffer", ""); - const stripped = input.trimStart(); - const openTag = OPEN_TAGS.find(tag => stripped.startsWith(tag)); + if (this.preBuffer) { + input = this.preBuffer + text; + this.replaceCarry("preBuffer", ""); + } else { + const stripped = text.trimStart(); + const leadingLength = text.length - stripped.length; + if (leadingLength > 0) this.appendPreWhitespace(text.slice(0, leadingLength)); + if (!stripped) return []; + input = stripped; + } + const openTag = OPEN_TAGS.find(tag => input.startsWith(tag)); if (openTag) { - const leadingLength = input.length - stripped.length; - const leading = this.interleaved ? input.slice(0, leadingLength) : ""; + const leading = this.finishPreWhitespace(this.interleaved); this.state = "thinking"; this.closeTag = closeTagFor(openTag); const events: AdapterEvent[] = leading ? [{ type: "text_delta", text: leading }] : []; - return this.drainChunk(input, leadingLength + openTag.length, events); + return this.drainChunk(input, openTag.length, events); } - if (stripped.length <= MAX_OPEN_TAG && isPossibleOpenTagPrefix(stripped)) { + if (input.length <= MAX_OPEN_TAG && isPossibleOpenTagPrefix(input)) { this.replaceCarry("preBuffer", input); return []; } this.state = "streaming"; - return input ? [{ type: "text_delta", text: input }] : []; + return [{ type: "text_delta", text: this.finishPreWhitespace(true) + input }]; } if (this.state === "thinking") { input = this.thinkingBuffer + text; @@ -102,8 +131,8 @@ export class InlineThinkTagParser { this.state = "streaming"; return out ? [{ type: "reasoning_raw_delta", text: out }] : []; } - if (this.preBuffer) { - const out = this.preBuffer; + if (this.preWhitespaceLength > 0 || this.preBuffer) { + const out = this.finishPreWhitespace(true) + this.preBuffer; this.replaceCarry("preBuffer", ""); this.state = "streaming"; return [{ type: "text_delta", text: out }]; @@ -113,6 +142,7 @@ export class InlineThinkTagParser { /** Release any partial tag/content carry when the owning stream stops early. */ dispose(): void { + this.finishPreWhitespace(false); this.replaceCarry("preBuffer", ""); this.replaceCarry("thinkingBuffer", ""); this.closeTag = ""; diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index 8ad48c0d4f1..0d30c3cd56f 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -64,6 +64,8 @@ inside subsequent code fences or quoted examples: this explicit opt-in does not Gateways producing ambiguous literals should use structured reasoning instead. A moving cursor scans each upstream chunk without copying the remaining response after every block. Only undecided leading input or a trailing tag fragment is retained and charged to the translator budget. +Undecided leading whitespace is charged one incoming segment at a time and joined only when +the initial format is decided or the stream ends. A block left unterminated at end of stream flushes as reasoning rather than being dropped. In interleaved Chat mode, whitespace after the leading block and after later blocks remains answer text, including indentation and blank lines. Kiro uses single-block mode and diff --git a/tests/adapters/openai/inline-think-boundaries.test.ts b/tests/adapters/openai/inline-think-boundaries.test.ts index c6411e13ad9..8e8fcf632d0 100644 --- a/tests/adapters/openai/inline-think-boundaries.test.ts +++ b/tests/adapters/openai/inline-think-boundaries.test.ts @@ -30,6 +30,8 @@ describe("inline thinking format boundaries", () => { test("Kiro retains single-block normalization and subsequent literal tags", () => { expect(split([" \nwhy\n answerliteral"], false)) .toEqual({ answer: "answerliteral", reasoning: "why" }); + expect(split([" ", "\n", "why\n answerliteral"], false)) + .toEqual({ answer: "answerliteral", reasoning: "why" }); }); test("ordinary code examples never activate parsing", () => { const input = "```xml\nliteral\n```"; @@ -62,7 +64,26 @@ describe("inline thinking format boundaries", () => { reserve.mockRestore(); } }); + test("split leading whitespace charges only newly retained bytes", () => { + const count = 20000; + const suffix = "ra"; + const budget = createTestTranslatorBudget({ maxTurnBytes: 100_000 }); + const reserve = spyOn(budget, "reserveTransient"); + const parser = new InlineThinkTagParser(budget, { interleaved: true }); + try { + for (let i = 0; i < count; i++) parser.feed(" "); + const events = [...parser.feed(suffix), ...parser.flush()]; + expect(projection(events)).toEqual({ answer: " ".repeat(count) + "a", reasoning: "r" }); + const reservedBytes = reserve.mock.calls.reduce((total, [bytes]) => total + bytes, 0); + expect(reservedBytes).toBeLessThanOrEqual(4 * Buffer.byteLength(" ".repeat(count) + suffix)); + } finally { + parser.dispose(); + reserve.mockRestore(); + } + expect(budget.snapshot().currentBytes).toBe(0); + }); test("partial tags and unterminated reasoning flush without loss", () => { + expect(split([" ", "\n"])).toEqual({ answer: " \n", reasoning: "" }); expect(split(["why😀xanswer Date: Wed, 23 Sep 2026 07:02:41 +0900 Subject: [PATCH 10/24] fix(meta-muse): consolidate login admission and bounded response handling (#5591) Carries #5591, which consolidates the closed #5234 and #5432, onto current dev. The provider contract keeps the inline-tag paragraph and adds the Meta Muse admission paragraph. Co-authored-by: Epinephrine --- .../src/content/docs/guides/providers.md | 24 ++--- .../docs/reference/platform-support.md | 17 ++-- src/oauth/meta-muse-device.ts | 56 ++++++++++-- src/server/management/oauth-account-routes.ts | 12 ++- structure/gui-and-management-api.md | 2 +- structure/providers-and-adapters.md | 9 +- tests/oauth/oauth-public-surface.test.ts | 90 +++++++++++++++++++ tests/providers/meta-muse-device.test.ts | 48 ++++++++++ 8 files changed, 227 insertions(+), 31 deletions(-) diff --git a/docs-site/src/content/docs/guides/providers.md b/docs-site/src/content/docs/guides/providers.md index e8cba6c928d..6258b8df496 100644 --- a/docs-site/src/content/docs/guides/providers.md +++ b/docs-site/src/content/docs/guides/providers.md @@ -135,7 +135,7 @@ Two exceptions are worth knowing because you can hit them: | Command Code | `ocx login command-code` — opencodex reads five-hour and weekly windows plus a credit balance | `commandcode` — the same service on `/provider/v1` with a key | | GitHub Copilot | `ocx login github-copilot` — requires an active Copilot subscription | the same `github-copilot` provider with `authMode: "key"`. The device flow above is the supported path, and either credential is a Copilot one, so the subscription still pays | | OrcaRouter | `ocx login orcarouter-oauth` — consent mints a user-owned, long-lived `sk-orca-…` key, and the request then carries a key | `orcarouter` — the same key pasted by hand | -| Meta Muse | `ocx login meta-muse` imports the Muse Code CLI key. Meta scopes that credential to its own CLI, so this is an unsupported use: how the calls settle is not observable from the API, and you should treat every call as billable against your account | `meta-model` is the supported path — every call is metered per token, and a Muse Code subscription does not work there | +| Meta Muse | `ocx login meta-muse` can import a local Muse Code CLI key or start device login. Meta scopes that credential to its own CLI, so this is an unsupported use: how the calls settle is not observable from the API, and you should treat every call as billable against your account | `meta-model` is the supported path — every call is metered per token, and a Muse Code subscription does not work there | Cursor, Kiro and Nous Portal are login-only and have no API-key equivalent. Google Antigravity is login-only too: `ocx login google-antigravity` signs in with your Google account over the Cloud Code @@ -697,18 +697,20 @@ material off it. Muse Spark is also reachable through resellers, with a narrower `command-code` carries both tiers, while `opencode-go` serves only `muse-spark-1.3-contributor`. -**Meta Muse Code (`meta-muse`).** On macOS, if you already use the Muse Code CLI, this -imports the API key it stored after `muse login` instead of asking you to provision a -second one. OpenCodex never launches the CLI: if no credential is present it tells you to -run `muse login` yourself. - -Elsewhere it asks you to paste the key. Meta ships no native Windows CLI, and on Linux the -CLI exists but where it stores its credential has not been verified, so OpenCodex refuses -to guess at a credential store and points you at [dev.meta.ai](https://dev.meta.ai) -instead, where the same key is visible. A pasted key faces the same format check and the -same live validation against the Model API as an imported one. See +**Meta Muse Code (`meta-muse`).** A plain macOS login first tries the API key already +stored by `muse login`. With no local credential, or on another platform, it starts the +browser device-approval flow. Add-account and reauthentication skip local import to avoid +reusing the account being replaced. OpenCodex never launches the Muse CLI. If device login +fails without cancellation, an available manual-input surface can accept a pasted key; +that key faces the same format and Model API validation as an imported key. See [Platform support](/reference/platform-support/) for the full per-platform picture. +Starting any Meta Muse login through the management API requires a dashboard session, +including add-account and reauthentication. A raw admin token or forged GUI headers receive +`403 oauth_consent_required` before a credential is read or a grant starts. This gate uses +the server-resolved session principal, not a separately recorded warning-checkbox receipt. +Direct `ocx login meta-muse` and other OAuth providers keep their existing login policies. + Both seeded `meta-muse` models expose `minimal`/`low`/`medium`/`high`/`xhigh`/`max` to routed clients, including Grok's effort picker. Requests use `User-Agent: muse-build/1.3.0 (opencodex compatibility)` so Meta accepts the Muse Code diff --git a/docs-site/src/content/docs/reference/platform-support.md b/docs-site/src/content/docs/reference/platform-support.md index a0d484551dc..29beee473a9 100644 --- a/docs-site/src/content/docs/reference/platform-support.md +++ b/docs-site/src/content/docs/reference/platform-support.md @@ -51,15 +51,13 @@ directly. ### Meta Muse Code -On macOS, OpenCodex imports the API key the Muse Code CLI already stored after -`muse login`, so you are not asked to provision a second one. - -Elsewhere it asks you to paste the key instead. Meta ships no native Windows -CLI, and on Linux the CLI exists but where it keeps its credential has not been -verified, so OpenCodex declines to guess at a credential store. The same key is -visible in [Meta's developer console](https://dev.meta.ai), and a pasted key -faces the same format check and the same live validation against the Model API -as an imported one. +On macOS, a plain login first tries the API key already stored by `muse login`. +If none is available, or on another platform, OpenCodex starts device approval +without launching the Muse CLI. Add-account and reauthentication skip import. +A failed, uncancelled device login can fall back to a manual key when the caller +offers an input surface. Pasted keys use the same format and Model API validation +as imported ones. Management login requires a dashboard session before either +credential-acquisition path; see the [provider guide](/guides/providers/). ## Windows notes @@ -78,4 +76,3 @@ OpenCodex states the actual reason rather than disabling a control silently. If a capability is unavailable on your platform, the error or the dashboard says which mechanism is missing and what the supported alternative is. If you hit one that does not, that is a bug worth reporting. - diff --git a/src/oauth/meta-muse-device.ts b/src/oauth/meta-muse-device.ts index 70ec7257821..fbff8b06124 100644 --- a/src/oauth/meta-muse-device.ts +++ b/src/oauth/meta-muse-device.ts @@ -23,6 +23,7 @@ */ import type { OAuthController, OAuthCredentials } from "./types"; import { sanitizeApiKeyValue } from "../providers/api-keys"; +import { BOUNDED_BODY_MAX_BYTES, readBoundedResponseBytes } from "../lib/bounded-body"; /** Meta's own Muse Code client. Public in its device-approval URL; not a secret. */ const CLIENT_ID = "1031625952748946"; @@ -173,12 +174,45 @@ function requestSignal(signal: AbortSignal | undefined): AbortSignal { return signal ? AbortSignal.any([signal, timeout]) : timeout; } +async function readMuseJson( + response: Response, + signal: AbortSignal, + kind: "device-authorization" | "device-token" | "mint-invalid", +): Promise | undefined> { + const declaredLength = response.headers.get("content-length"); + if (declaredLength && /^\d+$/.test(declaredLength) && Number(declaredLength) > BOUNDED_BODY_MAX_BYTES) { + void response.body?.cancel().catch(() => undefined); + throw new MuseDeviceLoginError( + kind, + `Muse Code response exceeded the ${BOUNDED_BODY_MAX_BYTES}-byte limit`, + { status: response.status }, + ); + } + const { bytes, oversized } = await readBoundedResponseBytes(response, { + maxBytes: BOUNDED_BODY_MAX_BYTES, + signal, + }); + if (oversized) { + throw new MuseDeviceLoginError( + kind, + `Muse Code response exceeded the ${BOUNDED_BODY_MAX_BYTES}-byte limit`, + { status: response.status }, + ); + } + try { + return record(JSON.parse(new TextDecoder("utf-8", { fatal: true }).decode(bytes))); + } catch { + return undefined; + } +} + /** Step 1: ask Meta for a user code. */ export async function requestMuseDeviceAuthorization( deps: MuseDeviceDeps = {}, signal?: AbortSignal, ): Promise { const now = deps.now ?? Date.now; + const request = requestSignal(signal); const response = await (deps.fetchImpl ?? fetch)(DEVICE_AUTHORIZATION_URL, { method: "POST", headers: { @@ -188,16 +222,19 @@ export async function requestMuseDeviceAuthorization( }, body: new URLSearchParams({ client_id: CLIENT_ID }).toString(), redirect: "error", - signal: requestSignal(signal), + signal: request, }); if (!response.ok) { + // The error body is never parsed, but it still has to be released: an unread + // response body keeps the underlying connection occupied. + void response.body?.cancel().catch(() => undefined); throw new MuseDeviceLoginError( "device-authorization", `Muse Code device authorization request failed: HTTP ${response.status}`, { status: response.status }, ); } - const payload = record(await response.json().catch(() => undefined)); + const payload = await readMuseJson(response, request, "device-authorization"); const deviceCode = text(payload?.device_code); const userCode = text(payload?.user_code); if (!deviceCode || !userCode) { @@ -241,6 +278,7 @@ export async function pollMuseDeviceToken( // shape checked the deadline at the top, so a sleep ending exactly at the deadline // skipped the final poll and discarded an approval the user had already completed // inside that window. + const request = requestSignal(signal); const response = await (deps.fetchImpl ?? fetch)(DEVICE_TOKEN_URL, { method: "POST", headers: { @@ -254,9 +292,9 @@ export async function pollMuseDeviceToken( grant_type: DEVICE_GRANT_TYPE, }).toString(), redirect: "error", - signal: requestSignal(signal), + signal: request, }); - const payload = record(await response.json().catch(() => undefined)); + const payload = await readMuseJson(response, request, "device-token"); if (response.ok) { // [W3] No deadline re-check here. If Meta answered 200 with a token, Meta accepted // the device code; its clock is authoritative and ours is not. Discarding an issued @@ -322,6 +360,7 @@ export async function mintMuseApiKey( signal?: AbortSignal, ): Promise { const now = deps.now ?? Date.now; + const request = requestSignal(signal); const response = await (deps.fetchImpl ?? fetch)(MUSE_KEY_URL, { method: "POST", headers: { @@ -332,9 +371,10 @@ export async function mintMuseApiKey( }, body: JSON.stringify(options.onboard ? { onboard: true } : {}), redirect: "error", - signal: requestSignal(signal), + signal: request, }); if (response.status === 429) { + void response.body?.cancel().catch(() => undefined); const wait = retryAfterMs(response.headers.get("retry-after"), now()); throw new MuseDeviceLoginError( "mint-rate-limited", @@ -345,14 +385,16 @@ export async function mintMuseApiKey( ); } if (!response.ok) { - // Status only. The body of this endpoint can carry the key itself. + // Status only. The body of this endpoint can carry the key itself, so it is + // never parsed — but it is still cancelled so the connection is released. + void response.body?.cancel().catch(() => undefined); throw new MuseDeviceLoginError( "mint-http", `Muse Code key exchange failed: HTTP ${response.status}`, { status: response.status }, ); } - const payload = record(await response.json().catch(() => undefined)); + const payload = await readMuseJson(response, request, "mint-invalid"); if (!payload) { throw new MuseDeviceLoginError("mint-invalid", "Muse Code key exchange returned an unreadable response", { status: response.status, diff --git a/src/server/management/oauth-account-routes.ts b/src/server/management/oauth-account-routes.ts index 8fcf66eba10..1d6379335fe 100644 --- a/src/server/management/oauth-account-routes.ts +++ b/src/server/management/oauth-account-routes.ts @@ -148,7 +148,7 @@ function validateKeyName( } export async function handleOauthAccountRoutes(ctx: ManagementContext): Promise { - const { req, url, config, deps, syncClaudeAgentDefsBestEffort } = ctx; + const { req, url, config, deps, principal, syncClaudeAgentDefsBestEffort } = ctx; if (url.pathname === "/api/accounts/events" && req.method === "GET") { const { accountSelectionStream } = await import("./account-selection-stream"); @@ -172,6 +172,16 @@ export async function handleOauthAccountRoutes(ctx: ManagementContext): Promise< const body = await readManagementJsonBodyOr(req, {}) as { provider?: string; addAccount?: boolean; accountId?: string; reauth?: boolean; openBrowser?: unknown }; const provider = (body.provider ?? "").trim().toLowerCase(); if (!isPublicOAuthProvider(provider)) return jsonResponse({ error: "unknown oauth provider" }, 400); + // Muse may import a local Keychain credential or start a device grant; add-account + // and reauth skip the import. All management login paths require the dashboard + // principal before credential acquisition. A raw token proves administration, + // not acknowledgement; caller-supplied headers are not consent evidence. + if (provider === "meta-muse" && principal !== "gui-session") { + return jsonResponse({ + error: "Meta Muse login requires acknowledgement in the OpenCodex dashboard.", + code: "oauth_consent_required", + }, 403); + } const namespaceCollision = codexAccountNamespaceProviderCollisionError(config.codexAccountNamespaces, provider); if (namespaceCollision) return jsonResponse({ error: namespaceCollision }, 409); const accountId = body.accountId?.trim(); diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index 9147b59e3a8..b2b13a5c274 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -161,7 +161,7 @@ this document owns is which module holds which area and what invariant that area | Updates | `GET /api/update/check`, `POST /api/update/run`, and `GET /api/update/status` own dashboard self-update state. A launched worker PID is persisted in `update-job.json`; dead PIDs recover immediately, while legacy active records without a PID recover only after ten minutes. Live PIDs remain exclusive regardless of record age. `GET /api/update/badge` backs the sidebar badge: it reports that an update exists and links to the update surface rather than gating other actions. | | Providers | Create/update/delete ordinary provider configs and enrich registry metadata. The reserved `openai` card exposes Pool(default)/Direct account mode; `openai-apikey` remains the separate API route. | | Models | Fetch routed model lists, disabled model visibility, and catalog-facing ids. New non-OAuth registration holds exposure until authoritative discovery; 20 or more distinct switch rows start OFF without disabling the provider. Pending rows cannot accept visibility changes. | -| OAuth | Login/status/logout for OAuth-backed providers, plus multiauth account management: `GET /api/oauth/accounts`, `PUT /api/oauth/accounts/active`, `PUT /api/oauth/accounts/alias`, `DELETE /api/oauth/accounts` list masked accounts per provider, switch the active one, edit its display-only alias, and remove one. The login flow itself is `GET /api/oauth/providers`, `POST /api/oauth/login`, `POST /api/oauth/login/code`, `POST /api/oauth/login/cancel`, `POST /api/oauth/logout`, and `GET /api/oauth/status`; pool controls are `GET/PUT/PATCH /api/oauth/accounts/pool` and `POST /api/oauth/accounts/clear-cooldown`. Login accepts `addAccount: true` to force a fresh browser identity. Device flows return a structured `deviceCode`; the GUI highlights and copies it before the user opens the verification page. | +| OAuth | Login/status/logout for OAuth-backed providers, plus multiauth account management: `GET /api/oauth/accounts`, `PUT /api/oauth/accounts/active`, `PUT /api/oauth/accounts/alias`, `DELETE /api/oauth/accounts` list masked accounts per provider, switch the active one, edit its display-only alias, and remove one. The login flow itself is `GET /api/oauth/providers`, `POST /api/oauth/login`, `POST /api/oauth/login/code`, `POST /api/oauth/login/cancel`, `POST /api/oauth/logout`, and `GET /api/oauth/status`; pool controls are `GET/PUT/PATCH /api/oauth/accounts/pool` and `POST /api/oauth/accounts/clear-cooldown`. Login accepts `addAccount: true` to force a fresh browser identity. Meta Muse login also requires the server-resolved `gui-session` principal before local import or device login (including reauth); see the [provider contract](providers-and-adapters.md). Device flows return a structured `deviceCode`; the GUI highlights and copies it before the user opens the verification page. | | Key providers | `GET /api/key-providers` exposes API-key provider presets for setup and dashboard flows, and `GET/POST/DELETE /api/keys` owns the proxy's own admission keys. Multi-key pool per key-auth provider: `GET /api/providers/keys`, `POST /api/providers/keys`, `PUT /api/providers/keys/active`, `PUT /api/providers/keys/alias`, `DELETE /api/providers/keys` masked list, add (upsert + activate), switch, rename, and remove keys. `provider.apiKey` always mirrors the active pool entry so routing stays single-key. | | OpenAI account mode | Report one OpenAI Codex card with Pool/Direct controls and one API-key card. Mode PATCH persists live without restart or catalog identity changes; Pool owns account/quota controls and Direct uses caller/main login only. Main-account DTOs report real credential presence and terminal `needsReauth` state instead of treating missing/invalid native auth as an unknown quota. Selection order has its own route: `PUT /api/codex-auth/accounts/priority` takes `{ id, priority }`, where `priority` is an integer -100..100 or `null` to restore the default, accepts `__main__`, 404s an unknown id, and echoes the stored value. Re-ordering never clears thread affinity, so the response carries no `appliesImmediately`, but it does release any pin — see [`openai-tiers.md`](providers/openai-tiers.md) for why. `PUT /api/codex-auth/active` with a null id releases one too, but that drops the operator's account selection along with it, so this route is the only operator-facing way to clear a pin while leaving the selected account in place. `GET /api/codex-auth/active` reports `pinned`, true only while the manually selected account is still the effective active one, plus `pinnedAccountId`, which names the pinned account whether or not it is the active one. Surfaces should render `pinnedAccountId`: under round-robin and fill-first the pin caps the tier ceiling at its own tier while the strategy cursor moves freely inside that tier, so `pinned` goes false on a sibling's turn even though the pin is still suppressing every higher tier — which is why the dashboard badges `pinnedAccountId` and the GUI controller tracks only the id. `pinned` answers the narrower question of whether routing is *currently* on the operator's choice; no surface in this repo asks it, and a new one almost certainly wants the id instead. | | Subagents | Read/write the featured `subagentModels` list capped at five ids. `GET/PUT /api/injection-model` manages the shared delegation model/effort selection, the independent OpenCodex guidance switch, and the default-off `syncCodexSubagentDefaults` opt-in for native Codex subagent defaults. When OpenCodex owns the active Codex routing, native `[agents]` defaults apply to newly created Codex tasks after sync/restart; external user-managed provider configs remain untouched. The defaults do not cause delegation and preserve existing user-owned defaults rather than overwriting them. PUT is partial-update: absent keys are unchanged, `null` clears, and non-object bodies are rejected with 400 before field validation. `syncCodexSubagentDefaults: true` requires a nonblank `model` and a supported Codex reasoning effort when effort is set; clearing `model` (null/empty) always clears effort and disables native-default sync even when the stored effort was invalid. | diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 82955de036d..6ed0c942dcf 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -3,6 +3,13 @@ The opt-in `inlineThinkTagModels` list follows static-policy override and model-rename rules; shared Kiro/Chat splitting and raw display follow [Chat compatibility](providers/chat-compat.md#inline-think-tag-recovery). +Meta Muse management login in `src/server/management/oauth-account-routes.ts` requires a +server-resolved `gui-session` before starting credential acquisition, including local import, +device login, add-account and reauthentication. This principal is not a checkbox receipt; +forged GUI headers and raw management credentials do not substitute for it. Direct CLI login +and other OAuth providers retain their existing policies. `src/oauth/meta-muse-device.ts` +cancels unparsed authorization/mint failures, including mint429, without reflecting their bodies. + OrcaRouter key exchange uses the shared raw-byte reader before returning a durable key. Its 64 KiB response ceiling, single 30-second header/body deadline, and cancellation behavior follow the [bounded ingestion contract](transports/inventory.md#bounded-response-ingestion-and-orcarouter-login). @@ -18,7 +25,7 @@ only canonical Fable, Opus, or Sonnet labels after removing terminal controls; u | `src/providers/model-rename-fields.ts`, `src/providers/model-rename-migration.ts` | Classifies every provider config field for a declared model rename. Exact-model records, lists and nested request-pacing keys follow the replacement; an already saved replacement entry wins. Provider-wide settings and credential fields are not model identities. | | `src/providers/resolved-model-policy.ts`, `src/providers/resolved-model-policy-merge.ts` | Static provider/model policy resolution for the final upstream wire model, plus its pure clone/merge/URL/family helpers. The resolver detaches and freezes registry defaults, operator overrides, exact explicit input-modality declarations, hard wire pins, aliases, and explicit false/empty values with field-level provenance. Provider derivation, routing, catalog hints, gather admission, and adapter selection consume its detached frozen result. Callers supply transport match, the exact capability row, and a credential-free effective auth decision; credential bytes, usability evidence, account/quota/health state, and observed limits remain outside the result. | -| `src/oauth/` | OAuth providers, token storage, refresh, and auth-token resolution. The login callback listener binds a per-provider FIXED loopback port, so consecutive logins reuse the same number; every response it sends ends its connection (`Connection: close`, including non-callback paths such as a stray `/favicon.ico` 404). Stopping the listener does not close an established socket, so without that a pooled client would deliver the next login's callback to the retired flow, which rejects the unknown state as a CSRF mismatch while the live flow waits. Kiro add-account identity prefers same-session `whoami` over a leftover SQLite state profile, and never persists the Builder ID service profile ARN as `accountId`. | +| `src/oauth/` | OAuth providers, token storage, refresh, and auth-token resolution. Meta Muse device authorization, polling, and key-mint JSON responses share the 64 KiB bounded-body ceiling and the request's deadline; oversized declared or streamed bodies are rejected before JSON parsing. The login callback listener binds a per-provider FIXED loopback port, so consecutive logins reuse the same number; every response it sends ends its connection (`Connection: close`, including non-callback paths such as a stray `/favicon.ico` 404). Stopping the listener does not close an established socket, so without that a pooled client would deliver the next login's callback to the retired flow, which rejects the unknown state as a CSRF mismatch while the live flow waits. Kiro add-account identity prefers same-session `whoami` over a leftover SQLite state profile, and never persists the Builder ID service profile ARN as `accountId`. | | `src/combos/request.ts` | Clones each selected combo target request and applies the existing target capability ladder: adaptive unknown targets and explicit empty ladders receive no unsupported reasoning/thinking controls, while known ladders retain per-target resolution. | | `src/adapters/openai-responses.ts` | Native OpenAI/ChatGPT Responses passthrough. | | `src/responses/muse-tool-name-alias.ts` | Host-gated Meta Muse 64-char tool-name alias/restore used by the Responses passthrough. | diff --git a/tests/oauth/oauth-public-surface.test.ts b/tests/oauth/oauth-public-surface.test.ts index fd9bb74a9eb..4937d4cddbc 100644 --- a/tests/oauth/oauth-public-surface.test.ts +++ b/tests/oauth/oauth-public-surface.test.ts @@ -18,6 +18,8 @@ import type { OcxConfig } from "../../src/types"; import type { OAuthController } from "../../src/oauth/types"; import { getCredential } from "../../src/oauth/store"; import * as oauthStore from "../../src/oauth/store"; +import * as oauth from "../../src/oauth"; +import { requestMuseDeviceAuthorization } from "../../src/oauth/meta-muse-device"; import { flushConfigDirHardeningForTests } from "../../src/config/paths"; import { setAsyncIcaclsRunnerForTests, setIcaclsRunnerForTests } from "../../src/lib/windows-secret-acl"; @@ -84,6 +86,94 @@ describe("legacy ChatGPT OAuth public-surface exclusion", () => { expect(isPublicOAuthProvider("github-copilot")).toBe(true); }); + test("Meta Muse login requires a consent-bearing GUI session", async () => { + const cfg = config(); + const request = () => new Request("http://localhost/api/oauth/login", { + method: "POST", + headers: { + "content-type": "application/json", + origin: "http://localhost", + "x-opencodex-gui-origin": "http://localhost", + "x-opencodex-csrf-token": "forgeable-without-a-session", + }, + // A missing account makes a correctly admitted request stop before the + // platform-specific import, while still proving it passed the consent gate. + body: JSON.stringify({ provider: "meta-muse", accountId: "missing-slot" }), + }); + + for (const principal of [undefined, "admin-token", "gui-pair-capability"] as const) { + const response = await handleManagementAPI(request(), new URL(request().url), cfg, {}, principal); + expect(response?.status).toBe(403); + expect(await response?.json()).toEqual({ + error: "Meta Muse login requires acknowledgement in the OpenCodex dashboard.", + code: "oauth_consent_required", + }); + } + + const admitted = await handleManagementAPI(request(), new URL(request().url), cfg, {}, "gui-session"); + expect(admitted?.status).toBe(404); + expect(await admitted?.json()).toEqual({ error: "Unknown account for reauth" }); + }); + + test.each([ + ["plain", {}, false], + ["add-account", { addAccount: true }, true], + ["reauth", { reauth: true }, true], + ] as const)("Muse %s admission precedes either credential-acquisition path", async (_mode, flags, forceLogin) => { + const cfg = config(); + saveConfig(cfg); + const login = spyOn(oauth, "startLoginFlow").mockResolvedValue({ url: "" }); + const request = (provider = "meta-muse") => new Request("http://localhost/api/oauth/login", { + method: "POST", + headers: { "content-type": "application/json", origin: "http://localhost", + "x-opencodex-gui-origin": "http://localhost", "x-opencodex-csrf-token": "forged" }, + body: JSON.stringify({ provider, ...flags, openBrowser: false }), + }); + try { + for (const principal of [undefined, "admin-token", "gui-pair-capability"] as const) { + const req = request(); + const response = await handleManagementAPI(req, new URL(req.url), cfg, {}, principal); + expect(response?.status).toBe(403); + expect((await response?.json())?.code).toBe("oauth_consent_required"); + } + expect(login).not.toHaveBeenCalled(); + const admitted = request(); + expect((await handleManagementAPI(admitted, new URL(admitted.url), cfg, {}, "gui-session"))?.status).toBe(200); + expect(login).toHaveBeenCalledWith("meta-muse", { forceLogin }, { onSettled: expect.any(Function) }); + const other = request("xai"); + expect((await handleManagementAPI(other, new URL(other.url), cfg, {}, "admin-token"))?.status).toBe(200); + expect(login).toHaveBeenLastCalledWith("xai", { forceLogin }, { onSettled: expect.any(Function) }); + } finally { login.mockRestore(); } + }); + + test("admitted Muse device overflow stays behind the public OAuth error boundary", async () => { + const cfg = config(); + saveConfig(cfg); + let fetches = 0; + const login = spyOn(oauth, "startLoginFlow").mockImplementation(async () => { + await requestMuseDeviceAuthorization({ fetchImpl: (async () => { + fetches++; + return new Response(JSON.stringify({ device_code: "private-device-canary", filler: "x".repeat(65_536) })); + }) as typeof fetch }); + throw new Error("oversized authorization must not succeed"); + }); + const request = () => new Request("http://localhost/api/oauth/login", { + method: "POST", headers: { "content-type": "application/json" }, + body: JSON.stringify({ provider: "meta-muse", addAccount: true, openBrowser: false }), + }); + try { + const denied = request(); + expect((await handleManagementAPI(denied, new URL(denied.url), cfg, {}, "admin-token"))?.status).toBe(403); + expect(fetches).toBe(0); + const admitted = request(); + const response = await handleManagementAPI(admitted, new URL(admitted.url), cfg, {}, "gui-session"); + expect(response?.status).toBe(409); + expect(await response?.json()).toEqual({ error: PUBLIC_OAUTH_ERROR }); + expect(fetches).toBe(1); + expect(getCredential("meta-muse")).toBeNull(); + } finally { login.mockRestore(); } + }); + test("generic management OAuth endpoints reject chatgpt before touching login state", async () => { const cfg = config(); const requests = [ diff --git a/tests/providers/meta-muse-device.test.ts b/tests/providers/meta-muse-device.test.ts index 6341efe050b..6d6fac8df3f 100644 --- a/tests/providers/meta-muse-device.test.ts +++ b/tests/providers/meta-muse-device.test.ts @@ -27,6 +27,7 @@ const KEY = `LLM|${"1".repeat(16)}|${"c".repeat(27)}`; const ACCOUNT_TOKEN = "meta-account-" + "z".repeat(48); /** If this string ever reaches an error message, a response body leaked into one. */ const BODY_CANARY = "canary-body-must-never-appear-in-an-error"; +const OVERSIZED_JSON = JSON.stringify({ value: "x".repeat(65_536) }); interface Reply { status?: number; body?: unknown; text?: string; headers?: Record } interface Scenario { @@ -132,6 +133,13 @@ describe("muse device authorization", () => { expect(error.message).toContain("500"); expect(error.message).not.toContain(BODY_CANARY); }); + + test("rejects an oversized streamed authorization response", async () => { + const h = harness({ auth: { text: OVERSIZED_JSON } }); + const error = await caught(() => requestMuseDeviceAuthorization(h.deps)); + expect(error.kind).toBe("device-authorization"); + expect(error.message).toContain("65536-byte limit"); + }); }); describe("muse device poll", () => { @@ -217,6 +225,17 @@ describe("muse device poll", () => { expect(error.kind).toBe("device-token"); }); + test("rejects an oversized token response instead of polling again", async () => { + const h = harness({ tokens: [{ text: OVERSIZED_JSON }] }); + const auth = await requestMuseDeviceAuthorization(h.deps); + const error = await caught(() => pollMuseDeviceToken(auth, h.deps)); + expect(error.kind).toBe("device-token"); + // A 200 without access_token also ends as device-token, so kind alone cannot tell + // the limit fired; the message names the bound. + expect(error.message).toContain("65536-byte limit"); + expect(h.calls.token).toBe(1); + }); + // W4 and W3 together: the last seconds of a grant must still be polled, and a token // the server issued in that window must not be thrown away by a local clock. test("polls once more inside the final seconds and accepts a late token", async () => { @@ -254,6 +273,23 @@ describe("muse device poll", () => { }); describe("muse key mint", () => { + test("cancels a rate-limited mint body without reading or reflecting it", async () => { + let cancelled = false; + let reads = 0; + const fetchImpl = (async () => new Response(new ReadableStream({ + pull(controller) { reads++; controller.enqueue(new TextEncoder().encode(BODY_CANARY)); }, + cancel() { cancelled = true; }, + }, { highWaterMark: 0 }), { status: 429, headers: { "retry-after": "30" } })) as typeof fetch; + const error = await caught(() => mintMuseApiKey(ACCOUNT_TOKEN, {}, { fetchImpl })); + expect(error.kind).toBe("mint-rate-limited"); + expect(error.status).toBe(429); + expect(error.retryAfterMs).toBe(30_000); + expect(error.message).toContain("30s"); + expect(error.message).not.toContain(BODY_CANARY); + expect(reads).toBe(0); + expect(cancelled).toBe(true); + }); + test("asks Meta to onboard during a login and sends the account bearer", async () => { const h = harness(); await mintMuseApiKey(ACCOUNT_TOKEN, { onboard: true }, h.deps); @@ -284,6 +320,18 @@ describe("muse key mint", () => { expect(error.message).not.toContain(BODY_CANARY); }); + test("rejects an oversized declared mint response before consuming it", async () => { + let cancelled = false; + const fetchImpl = (async () => new Response(new ReadableStream({ + pull() {}, + cancel() { cancelled = true; }, + }), { headers: { "content-length": "65537" } })) as typeof fetch; + const error = await caught(() => mintMuseApiKey(ACCOUNT_TOKEN, {}, { fetchImpl })); + expect(error.kind).toBe("mint-invalid"); + expect(error.message).toContain("65536-byte limit"); + expect(cancelled).toBeTrue(); + }); + test("lowercases the email and keeps the usage object", async () => { const h = harness({ mint: { body: { ...MINT_OK, subs_usage: { weekly: { used_percent: 4 } } } } }); const payload = await mintMuseApiKey(ACCOUNT_TOKEN, {}, h.deps); From ae21ec07435529d186e51a9fdc9498b6c9f99252 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Wed, 23 Sep 2026 07:02:41 +0900 Subject: [PATCH 11/24] fix(claude-desktop): keep applied state consistent across profile edits (#5590) Carries #5590, which consolidates the closed #5337, onto current dev. Co-authored-by: Epinephrine Co-authored-by: luvs01 --- src/claude/desktop-profile.ts | 34 ++- src/cli/claude-desktop.ts | 5 +- src/cli/claude.ts | 4 +- .../management/agent-settings-routes.ts | 13 +- src/server/management/config-routes.ts | 40 ++- structure/clients/claude-desktop.md | 9 + tests/claude-integration/claude-cli.test.ts | 4 +- .../claude-desktop-cli.test.ts | 41 ++- .../claude-management-api.test.ts | 32 +++ tests/clients/desktop-profile.test.ts | 33 +-- .../clients/sync-client-integrations.test.ts | 238 +++++++++++++++++- 11 files changed, 404 insertions(+), 49 deletions(-) diff --git a/src/claude/desktop-profile.ts b/src/claude/desktop-profile.ts index e35b74cb464..c44e7ce0246 100644 --- a/src/claude/desktop-profile.ts +++ b/src/claude/desktop-profile.ts @@ -77,15 +77,6 @@ function assertExactKeys(value: Record, keys: readonly string[] } } -/** - * Applied-state markers survive every profile rebuild. - * - * `parseDesktopProfile`, `reconcileDesktopProfile` and `moveDesktopRoute` each construct a - * fresh `{ version, assignments, defaults }`, and the management routes persist whatever they - * return. Without this carry-through, saving an assignment — or merely dragging a model to - * another family — would erase the fingerprint the apply route wrote, and the GUI would report - * "not applied" for a config that is applied on disk. - */ function appliedMarkers(source: { appliedFingerprint?: unknown; appliedAt?: unknown }): { appliedFingerprint?: string; appliedAt?: string; @@ -96,6 +87,19 @@ function appliedMarkers(source: { appliedFingerprint?: unknown; appliedAt?: unkn }; } +export function sameProfileContent(left: DesktopProfile, right: DesktopProfile): boolean { + return DESKTOP_FAMILIES.every(family => left.defaults[family] === right.defaults[family]) + && JSON.stringify(Object.entries(left.assignments).sort(([a], [b]) => a.localeCompare(b))) + === JSON.stringify(Object.entries(right.assignments).sort(([a], [b]) => a.localeCompare(b))); +} + +/** Retain applied-state bookkeeping only when the desired Desktop config is unchanged. */ +export function preserveDesktopAppliedState(source: DesktopProfile, rebuilt: DesktopProfile): DesktopProfile { + return sameProfileContent(source, rebuilt) + ? { ...rebuilt, ...appliedMarkers(source) } + : rebuilt; +} + function isFamily(value: unknown): value is DesktopFamily { return typeof value === "string" && (DESKTOP_FAMILIES as readonly string[]).includes(value); } @@ -244,7 +248,8 @@ export function reconcileDesktopProfile( const current = defaults[family]; defaults[family] = current && assignments[current]?.family === family ? current : (members[0] ?? null); } - return parseDesktopProfile({ version: 1, assignments, defaults, ...appliedMarkers(profile) }); + const rebuilt = parseDesktopProfile({ version: 1, assignments, defaults }); + return preserveDesktopAppliedState(profile, rebuilt); } export function moveDesktopRoute( @@ -268,7 +273,7 @@ export function moveDesktopRoute( const destinationMembers = Object.keys(assignments).filter(key => assignments[key]!.family === family).sort(); if (makeDefault || !defaults[family] || assignments[defaults[family]!]?.family !== family) defaults[family] = route; if (!defaults[family] && destinationMembers.length > 0) defaults[family] = destinationMembers[0]!; - return parseDesktopProfile({ version: 1, assignments, defaults, ...appliedMarkers(parsed) }); + return parseDesktopProfile({ version: 1, assignments, defaults }); } export function setDesktopFamilyDefault( @@ -280,7 +285,12 @@ export function setDesktopFamilyDefault( const members = Object.keys(parsed.assignments).filter(key => parsed.assignments[key]!.family === family); if (route === null && members.length > 0) throw new DesktopProfileError("cannot clear a non-empty family default", `profile.defaults.${family}`); if (route !== null && parsed.assignments[route]?.family !== family) throw new DesktopProfileError("route is not a member of this family", `profile.defaults.${family}`); - return parseDesktopProfile({ ...parsed, defaults: { ...parsed.defaults, [family]: route } }); + const rebuilt = parseDesktopProfile({ + version: 1, + assignments: parsed.assignments, + defaults: { ...parsed.defaults, [family]: route }, + }); + return preserveDesktopAppliedState(parsed, rebuilt); } export function renderDesktopProfile( diff --git a/src/cli/claude-desktop.ts b/src/cli/claude-desktop.ts index 5baaec6f995..e637ba00d96 100644 --- a/src/cli/claude-desktop.ts +++ b/src/cli/claude-desktop.ts @@ -398,7 +398,8 @@ export async function handleClaudeDesktopCommand(argv: string[], deps: ApplyProf const applyInvocation = argv.length === 0 || command === "apply" || applyFlags.length > 0; if (applyInvocation) { const rest = argv.filter(arg => arg !== "apply"); - const parsedTarget = parseDesktopApplyArgs(rest, loadConfig()); + const preApplyConfig = loadConfig(); + const parsedTarget = parseDesktopApplyArgs(rest, preApplyConfig); if ("error" in parsedTarget) { console.error(parsedTarget.error); return 2; } const { target } = parsedTarget; try { @@ -424,7 +425,7 @@ export async function handleClaudeDesktopCommand(argv: string[], deps: ApplyProf console.log(`Claude Desktop gateway 설정을 적용했습니다: ${result.path}`); for (const line of gatewayModeExplanation({ requestedExplicitly: applyFlags.some(flag => flag !== "--first-party"), - config: loadConfig(), + config: preApplyConfig, })) { console.log(line); } diff --git a/src/cli/claude.ts b/src/cli/claude.ts index 3daec5744c9..846a8c921e6 100644 --- a/src/cli/claude.ts +++ b/src/cli/claude.ts @@ -572,7 +572,6 @@ export function claudeLaunchPreflight( */ const NATIVE_STRIPPED_LEVERS = [ "CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY", - "CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST", "CLAUDE_CODE_MAX_CONTEXT_TOKENS", "CLAUDE_CODE_AUTO_COMPACT_WINDOW", "CLAUDE_CODE_ALWAYS_ENABLE_EFFORT", @@ -632,6 +631,9 @@ export function buildNativeClaudeEnv( } for (const name of NATIVE_STRIPPED_LEVERS) delete env[name]; + // An explicit caller-owned guard must follow a caller-owned gateway and credential; + // otherwise settings.env can replace the destination while retaining the credential. + if (hasOwnedAdmission) delete env.CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST; const providerNames = Object.keys(config.providers); for (const name of MODEL_ENV_SLOT_NAMES) { const value = env[name]; diff --git a/src/server/management/agent-settings-routes.ts b/src/server/management/agent-settings-routes.ts index d15b5117e3b..845a754474c 100644 --- a/src/server/management/agent-settings-routes.ts +++ b/src/server/management/agent-settings-routes.ts @@ -985,7 +985,7 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise let body: { profile?: unknown }; try { body = await readManagementJsonBody(req); } catch (error) { rethrowManagementBodyTooLarge(error); return jsonResponse({ error: "invalid JSON body" }, 400); } try { - const { parseDesktopProfile, reconcileDesktopProfile } = await import("../../claude/desktop-profile"); + const { parseDesktopProfile, preserveDesktopAppliedState, reconcileDesktopProfile } = await import("../../claude/desktop-profile"); const parsed = parseDesktopProfile(body.profile); const current = await buildClaudeDesktopState(config); const availableRoutes = new Set(current.models.filter(item => item.available).map(item => item.route)); @@ -1008,8 +1008,15 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise throw new Error(`현재 사용할 수 없는 모델은 기본값으로 지정할 수 없습니다: ${nextDefault}`); } } - const state = await buildClaudeDesktopState(config, parsed); - config.claudeCode = { ...(config.claudeCode ?? {}), desktopProfile: reconcileDesktopProfile(state.profile, state.models) }; + // Applied markers are server-owned bookkeeping. Discard client copies, then + // restore the trusted markers only if the desired profile stayed identical. + const editable = { version: 1 as const, assignments: parsed.assignments, defaults: parsed.defaults }; + const state = await buildClaudeDesktopState(config, editable); + const rebuilt = reconcileDesktopProfile(state.profile, state.models); + config.claudeCode = { + ...(config.claudeCode ?? {}), + desktopProfile: preserveDesktopAppliedState(current.profile, rebuilt), + }; saveConfigPreservingClaudeCode(config); const saved = await buildClaudeDesktopState(config); const runtimePort = Number(url.port) || config.port; diff --git a/src/server/management/config-routes.ts b/src/server/management/config-routes.ts index 6db5ca3f35c..77e7dec61f0 100644 --- a/src/server/management/config-routes.ts +++ b/src/server/management/config-routes.ts @@ -18,6 +18,7 @@ import { isValidProviderName, loadConfig, multiAgentGuidanceEnabled, + mutatePersistedConfig, providerBaseUrlConfigError, providerHeadersConfigError, saveConfigPreservingClaudeCode, @@ -228,9 +229,42 @@ export async function syncEnabledClientIntegrations( latest.claudeCode?.desktopProfile, nativeContextLimits(latest), ); - out.push(r.written - ? { client: "claude-desktop", ok: true, changed: true } - : { client: "claude-desktop", ok: false, reason: r.reason ?? "Claude Desktop write failed" }); + if (!r.written || !r.fingerprint) { + out.push({ client: "claude-desktop", ok: false, reason: r.reason ?? "Claude Desktop write failed" }); + } else { + const { emptyDesktopProfile, sameProfileContent } = await import("../../claude/desktop-profile"); + // The fingerprint belongs to the desired profile the write just used. If another + // writer saved a different desired profile between the Desktop write and this marker + // commit, stamping it would claim B is applied while the disk holds A's bytes. + const writtenProfile = latest.claudeCode?.desktopProfile; + const marked = mutatePersistedConfig(persisted => { + const profile = persisted.claudeCode?.desktopProfile; + // Presence first: a concurrent delete (of the profile or the whole claudeCode + // subtree) must not resurrect the written profile under a fresh fingerprint, + // and a concurrent insert must not inherit it either. Content is compared only + // when both sides carry a profile. + if ((profile == null) !== (writtenProfile == null)) { + return { changed: false, value: false }; + } + if (profile && writtenProfile && !sameProfileContent(profile, writtenProfile)) { + return { changed: false, value: false }; + } + persisted.claudeCode = { + ...(persisted.claudeCode ?? {}), + desktopProfile: { + ...(profile ?? writtenProfile ?? emptyDesktopProfile()), + appliedFingerprint: r.fingerprint, + appliedAt: new Date().toISOString(), + }, + }; + return { changed: true, value: true }; + }); + out.push(marked.status === "unavailable" + ? { client: "claude-desktop", ok: false, reason: "Claude Desktop applied marker was not saved (" + marked.reason + ")" } + : marked.value === false + ? { client: "claude-desktop", ok: false, reason: "Claude Desktop desired profile changed during sync; applied marker skipped" } + : { client: "claude-desktop", ok: true, changed: true }); + } } } catch (error) { out.push({ client: "claude-desktop", ok: false, reason: error instanceof Error ? error.message : String(error) }); diff --git a/structure/clients/claude-desktop.md b/structure/clients/claude-desktop.md index e313fa2fcc6..656855e6002 100644 --- a/structure/clients/claude-desktop.md +++ b/structure/clients/claude-desktop.md @@ -151,6 +151,15 @@ restores Desktop even with `--keep-catalog`; retries preserve the original catal not clear a newer connection. Authorized uninstall completes or resumes owned Desktop cleanup before removing OpenCodex state, and preserves recovery state when cleanup conflicts or fails. +The server-owned applied marker (`claudeCode.desktopProfile.appliedFingerprint` and +`appliedAt`) is committed by the config route only while the persisted desired profile still +matches the profile that was just written. Presence is compared first, then content: if a concurrent +writer deleted the profile or the whole `claudeCode` block, or replaced it with a different +profile, before the marker commit, the write is declined and reported as skipped instead of +resurrecting the removed profile with a fresh fingerprint. A profile that was absent from the start +still stores its fingerprint normally. Default-family key order does not change the desired content; +the comparison uses each family's selected route while preserving real selection changes. + These guarantees concern files on disk. Fully quitting and reopening Desktop is required after apply, rotation/recovery or restoration; there is no automatic process restart or guarantee that a running app discarded a key. Local disconnect does not revoke the hub key or remove arbitrary diff --git a/tests/claude-integration/claude-cli.test.ts b/tests/claude-integration/claude-cli.test.ts index 900ebea069c..6e73cc8ed0f 100644 --- a/tests/claude-integration/claude-cli.test.ts +++ b/tests/claude-integration/claude-cli.test.ts @@ -126,17 +126,19 @@ describe("ocx claude native fallback", () => { expect(env.CLAUDE_CODE_AUTO_COMPACT_WINDOW).toBeUndefined(); }); - test("preserves an unrelated loopback gateway and its user credential", () => { + test("preserves an unrelated gateway, its user credential, and its host-managed guard", () => { for (const baseUrl of ["http://localhost:8080", "http://127.0.0.1:10100"]) { const env = buildNativeClaudeEnv(cfg({ port: 10100 }), { ANTHROPIC_BASE_URL: baseUrl, ANTHROPIC_API_KEY: "sk-ant-user-key", + CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST: "1", }, { preBunAnthropicSlots: ["ANTHROPIC_BASE_URL", "ANTHROPIC_API_KEY"], }); expect(env.ANTHROPIC_BASE_URL).toBe(baseUrl); expect(env.ANTHROPIC_API_KEY).toBe("sk-ant-user-key"); + expect(env.CLAUDE_CODE_PROVIDER_MANAGED_BY_HOST).toBe("1"); } }); diff --git a/tests/claude-integration/claude-desktop-cli.test.ts b/tests/claude-integration/claude-desktop-cli.test.ts index a33026bf349..77d74d6eca5 100644 --- a/tests/claude-integration/claude-desktop-cli.test.ts +++ b/tests/claude-integration/claude-desktop-cli.test.ts @@ -1,7 +1,7 @@ import { afterEach, beforeEach, expect, spyOn, test } from "bun:test"; import { existsSync, mkdirSync, mkdtempSync, readFileSync, unlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; -import { join } from "node:path"; +import { dirname, join } from "node:path"; import { applyProfile as applyProfileProduction, handleClaudeDesktopCommand as handleClaudeDesktopCommandProduction, type ApplyProfileDeps } from "../../src/cli/claude-desktop"; import * as managementApi from "../../src/server/management-api"; import { buildClaudeDesktopState } from "../../src/server/management-api"; @@ -12,6 +12,8 @@ import * as lifecycleLock from "../../src/client/lifecycle-lock"; import { readClientConnectionState, clearClientConnection } from "../../src/client/state"; import { HubClientError } from "../../src/client/hub-client"; import { claudeDesktopIntegrationEnabledNow, setIntegrationEnabled } from "../../src/codex/desired-state"; +import { resetCodexRuntimeResolveCacheForTests, setCodexRuntimeResolveCacheForTests } from "../../src/codex/runtime"; +import { resetBundledCatalogCacheForTests, setBundledCatalogCacheForTests } from "../../src/codex/catalog/bundled"; import { serviceApiTokenBackupPath, serviceApiTokenFilePath, writeServiceApiTokenFile } from "../../src/lib/service-secrets"; import type { OcxConfig } from "../../src/types"; import { removeTreeWithRetry } from "../helpers/remove-tree"; @@ -27,13 +29,30 @@ const applyProfile = (profile: Parameters[0], mod const handleClaudeDesktopCommand = (args: string[], deps: ApplyProfileDeps = {}) => handleClaudeDesktopCommandProduction(args, { lifecycleLockDeps: fixtureLock(), ...deps }); +// Fixture-stage config placement only: the verified writers under test still run +// saveConfig, but arranging a fixture through it pays the mutation-lock and ACL +// subprocess cost (~0.5-1s on Windows) for state no assertion inspects. +function writeFixtureConfig(config: OcxConfig): void { + const path = getConfigPath(); + mkdirSync(dirname(path), { recursive: true }); + writeFileSync(path, JSON.stringify(config), { mode: 0o600 }); +} + beforeEach(() => { previousHome = process.env.OPENCODEX_HOME; previousDesktopDir = process.env.OPENCODEX_CLAUDE_DESKTOP_CONFIG_DIR; dir = mkdtempSync(join(tmpdir(), "ocx-desktop-cli-")); process.env.OPENCODEX_HOME = join(dir, "ocx"); process.env.OPENCODEX_CLAUDE_DESKTOP_CONFIG_DIR = join(dir, "desktop"); - saveConfig({ + // Keep host Codex work out of the fixture: a real runtime probe plus the + // bundled-catalog subprocess cost ~1s per buildClaudeDesktopState call on this + // path, while the tests only need a deterministic catalog projection. + setCodexRuntimeResolveCacheForTests( + { runtime: { command: "codex", version: null, source: "fallback" }, failures: [] }, + { discoverAlternatives: false }, + ); + setBundledCatalogCacheForTests({ command: "codex", version: null }, null); + writeFixtureConfig({ port: 10100, defaultProvider: "mock", providers: { @@ -45,6 +64,8 @@ beforeEach(() => { afterEach(() => { restoreLocalBuild?.(); restoreLocalBuild = undefined; + resetBundledCatalogCacheForTests(); + resetCodexRuntimeResolveCacheForTests(); if (previousHome === undefined) delete process.env.OPENCODEX_HOME; else process.env.OPENCODEX_HOME = previousHome; if (previousDesktopDir === undefined) delete process.env.OPENCODEX_CLAUDE_DESKTOP_CONFIG_DIR; @@ -66,7 +87,7 @@ function connectDesktopFixture(blockLocalBuild = true): void { selectedClients: ["codex"], tokenEnv: "OPENCODEX_API_AUTH_TOKEN", apiKeyId: "desktop-key", tokenFingerprint: fingerprint, protocolVersion: 1, connectedAt: "2026-09-06T00:00:00.000Z", }; - saveConfig(config); + writeFixtureConfig(config); expect(readClientConnectionState().kind).toBe("connected"); if (blockLocalBuild) { const spy = spyOn(managementApi, "buildClaudeDesktopState").mockImplementation(async () => { @@ -91,7 +112,8 @@ test.each([ ["--static", "static"], ["--hybrid", "hybrid"], ["--discovery-only", "discovery"], ] as const)("connected CLI %s applies exact hub IDs without local reconciliation", async (flag, mode) => { connectDesktopFixture(); - setIntegrationEnabled("claude-desktop", false); + // Fixture placement of the disabled switch; apply itself must flip it back on. + writeFixtureConfig({ ...loadConfig(), clientIntegrations: { "claude-desktop": false } }); const log = spyOn(console, "log").mockImplementation(() => {}); const warn = spyOn(console, "warn").mockImplementation(() => {}); const error = spyOn(console, "error").mockImplementation(() => {}); @@ -507,14 +529,19 @@ test("apply writes locally only when no proxy is running", async () => { expect(existsSync(join(dir, "desktop"))).toBe(true); }); -test("no-arg and legacy mode flags apply Desktop config", async () => { +test.each([{ args: [] as string[] }, { args: ["--static"] }])("no-arg and legacy mode flags apply Desktop config: $args", async ({ args }) => { + const config = loadConfig(); + config.claudeCode = { intercept: { enabled: false } }; + writeFixtureConfig(config); const log = spyOn(console, "log").mockImplementation(() => {}); const error = spyOn(console, "error").mockImplementation(() => {}); try { // Deterministic: no live proxy in the test environment, so apply writes locally. const noProxy = { findLiveProxyImpl: async () => null }; - expect(await handleClaudeDesktopCommand([], noProxy)).toBe(0); - expect(await handleClaudeDesktopCommand(["--static"], noProxy)).toBe(0); + expect(await handleClaudeDesktopCommand(args, noProxy)).toBe(0); + if (args.length === 0) { + expect(log.mock.calls.flat().join(" ")).not.toContain("ocx claude desktop apply --first-party"); + } expect(readFileSync(join(process.env.OPENCODEX_CLAUDE_DESKTOP_CONFIG_DIR!, "_meta.json"), "utf8")).toContain("opencodex"); expect(error).not.toHaveBeenCalled(); } finally { diff --git a/tests/claude-integration/claude-management-api.test.ts b/tests/claude-integration/claude-management-api.test.ts index 751b838ee99..8c4a7994b16 100644 --- a/tests/claude-integration/claude-management-api.test.ts +++ b/tests/claude-integration/claude-management-api.test.ts @@ -875,6 +875,38 @@ test("Claude Desktop PUT rejects invalid JSON profile without mutating saved con } }); +test("Claude Desktop PUT clears applied markers when routing changes", async () => { + const server = startServer(0); + try { + const apply = await fetch(new URL("/api/claude-desktop/apply", server.url), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ mode: "static" }), + }); + expect(apply.status).toBe(200); + expect(loadConfig().claudeCode?.desktopProfile?.appliedFingerprint).toBeString(); + + const state = await fetch(new URL("/api/claude-desktop", server.url)).then(r => r.json()) as Record; + const edited = structuredClone(state.profile); + edited.assignments["mock/test-model"].family = "sonnet"; + edited.defaults.opus = Object.keys(edited.assignments) + .filter(route => edited.assignments[route].family === "opus") + .sort()[0] ?? null; + edited.defaults.sonnet = "mock/test-model"; + + const put = await fetch(new URL("/api/claude-desktop", server.url), { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ profile: edited }), + }); + expect(put.status).toBe(200); + expect(loadConfig().claudeCode?.desktopProfile).not.toHaveProperty("appliedFingerprint"); + expect(loadConfig().claudeCode?.desktopProfile).not.toHaveProperty("appliedAt"); + } finally { + await server.stop(true); + } +}); + test("Claude Desktop PUT retains but cannot move an unavailable route", async () => { const seeded = loadConfig(); seeded.claudeCode = { diff --git a/tests/clients/desktop-profile.test.ts b/tests/clients/desktop-profile.test.ts index 7aa7cdda7e1..82725d65a70 100644 --- a/tests/clients/desktop-profile.test.ts +++ b/tests/clients/desktop-profile.test.ts @@ -120,10 +120,8 @@ describe("Claude Desktop profile", () => { }); // The apply route writes `appliedFingerprint`/`appliedAt` back onto the stored profile so the - // GUI can show applied-vs-saved state. Every rebuild in this module must accept AND carry them: - // rejecting them broke the Desktop tab outright after the first apply, and silently dropping - // them would make a saved edit — or a single drag between families — report "not applied" for a - // config that is applied on disk. + // GUI can show applied-vs-saved state. Parsing and no-op rebuilds retain them, while a change to + // the desired Desktop config must clear them so the old on-disk config is not reported as current. describe("applied-state markers", () => { const applied = { appliedFingerprint: "0123456789abcdef", @@ -140,23 +138,28 @@ describe("Claude Desktop profile", () => { expect(parsed.appliedAt).toBe(applied.appliedAt); }); - test("reconcileDesktopProfile keeps them across a catalog change", () => { + test("reconcileDesktopProfile clears them across a catalog change", () => { const next = reconcileDesktopProfile(seeded(), [...models, { route: "test/new-model", label: "New" }]); - expect(next.appliedFingerprint).toBe(applied.appliedFingerprint); - expect(next.appliedAt).toBe(applied.appliedAt); + expect(next).not.toHaveProperty("appliedFingerprint"); + expect(next).not.toHaveProperty("appliedAt"); }); - test("moveDesktopRoute keeps them — the drag-and-drop path", () => { - const moved = moveDesktopRoute(seeded(), "cursor/gpt-5.6-luna", "sonnet"); - expect(moved.appliedFingerprint).toBe(applied.appliedFingerprint); - expect(moved.appliedAt).toBe(applied.appliedAt); + test("reconcileDesktopProfile keeps them when profile content is unchanged", () => { + expect(reconcileDesktopProfile(seeded(), models)).toMatchObject(applied); }); - test("setDesktopFamilyDefault keeps them", () => { + test("moveDesktopRoute clears them — the drag-and-drop path", () => { const moved = moveDesktopRoute(seeded(), "cursor/gpt-5.6-luna", "sonnet"); - const next = setDesktopFamilyDefault(moved, "sonnet", "cursor/gpt-5.6-luna"); - expect(next.appliedFingerprint).toBe(applied.appliedFingerprint); - expect(next.appliedAt).toBe(applied.appliedAt); + expect(moved).not.toHaveProperty("appliedFingerprint"); + expect(moved).not.toHaveProperty("appliedAt"); + }); + + test("setDesktopFamilyDefault clears them only when the default changes", () => { + const changed = setDesktopFamilyDefault(seeded(), "opus", "native/gpt-5.6-sol"); + expect(changed).not.toHaveProperty("appliedFingerprint"); + expect(changed).not.toHaveProperty("appliedAt"); + expect(setDesktopFamilyDefault(seeded(), "opus", "anthropic/claude-fable-5")) + .toMatchObject(applied); }); test("a profile without the markers stays without them", () => { diff --git a/tests/clients/sync-client-integrations.test.ts b/tests/clients/sync-client-integrations.test.ts index 219a1480829..3f6b4b4bcb2 100644 --- a/tests/clients/sync-client-integrations.test.ts +++ b/tests/clients/sync-client-integrations.test.ts @@ -69,11 +69,12 @@ describe("ocx sync fans out to enabled native clients and owned file integration expect(fn).toContain("refreshOwnedCatalogIntegrations"); // Native clients keep their catches; the owned catalog helper isolates file clients. expect(fn.match(/catch \(error\)/g)?.length).toBe(2); - // The Desktop write gets the native context limits, same as every other Desktop - // call site. 8b672205e threaded `nativeContextLimits` through those writers and - // left this assertion naming the retired `providerContextCap` spelling, so the - // source-shape check failed against the very change it is meant to pin. expect(fn).toContain("nativeContextLimits(latest)"); + // Cleanup accepts only the fingerprint of the exact credential-bearing profile we wrote. + // Sync must durably advance that ownership marker rather than leaving the old value behind. + expect(fn).toContain("mutatePersistedConfig(persisted =>"); + expect(fn).toContain("appliedFingerprint: r.fingerprint"); + expect(fn.indexOf("nativeContextLimits(latest)")).toBeLessThan(fn.indexOf("appliedFingerprint: r.fingerprint")); // A client that is off is omitted rather than reported: the caller has to be able to // tell "left alone" from "tried and failed", so there is no skipped state to emit. expect(fn).not.toContain('"skipped"'); @@ -144,7 +145,7 @@ describe("Desktop sync rechecks persisted state after discovery", () => { writes.push(args); return outcome === "refusal" ? { written: false, path: "fixture", reason: "desktop_remote_store_active" } - : { written: true, path: "fixture" }; + : { written: true, path: "fixture", fingerprint: "0123456789abcdef" }; }, }); try { @@ -201,6 +202,233 @@ describe("Desktop sync rechecks persisted state after discovery", () => { } }); } + test("a desired-profile change during the Desktop write keeps the new profile without the old fingerprint", async () => { + // The marker commit must not stamp the fingerprint of the profile whose bytes were + // written (A) onto a different desired profile (B) saved by a concurrent writer while + // the Desktop write was in flight. B stays persisted and the sync reports the skip. + const profileA = { + version: 1 as const, + assignments: { "mock/hidden": { family: "opus" as const, alias: "claude-opus-4-8-20260201" } }, + defaults: { opus: "mock/hidden", fable: null, sonnet: null, haiku: null }, + }; + const profileB = { + version: 1 as const, + assignments: { "mock/keep": { family: "sonnet" as const, alias: "claude-opus-4-8-20260202" } }, + defaults: { opus: null, fable: null, sonnet: "mock/keep", haiku: null }, + }; + const config: OcxConfig = { + port: 10100, + defaultProvider: "mock", + clientIntegrations: { grok: false }, + providers: { + mock: { adapter: "openai-chat", baseUrl: "https://example.test/v1", models: ["keep", "hidden"] }, + openai: { adapter: "openai-responses", baseUrl: "https://example.test/v1", contextWindow: 400_000 }, + }, + apiKeys: [{ id: "sync-key", name: "fixture", key: "ocx_old_sync_fixture", createdAt: "2026-01-01T00:00:00.000Z" }], + claudeCode: { desktopProfile: profileA }, + }; + writeFileSync(join(root, "config.json"), JSON.stringify(config)); + const models: CatalogModel[] = [ + { provider: "mock", id: "keep", contextWindow: 123_000 }, + { provider: "mock", id: "hidden", contextWindow: 456_000 }, + ]; + const writes: Parameters[] = []; + const realRefresh = ownedRefresh.refreshOwnedIntegration; + const refresh = spyOn(ownedRefresh, "refreshOwnedIntegration").mockImplementation((input, options) => + input.clientId === "mcode" + ? Promise.resolve({ client: "mcode", ok: true, changed: true }) + : realRefresh(input, options)); + const aside = spyOn(asideProfiles, "refreshAsideProfiles").mockResolvedValue([]); + try { + const results = await syncEnabledClientIntegrations(12345, config, { + fetchAllModels: async () => models, + writeDesktop3pConfig: (...args) => { + writes.push(args); + // A concurrent writer saves desired profile B while the Desktop write is in flight. + const drifted = structuredClone(config); + drifted.claudeCode = { desktopProfile: profileB }; + writeFileSync(join(root, "config.json"), JSON.stringify(drifted)); + return { written: true, path: "fixture", fingerprint: "0123456789abcdef" }; + }, + }); + expect(writes).toHaveLength(1); + const outcome = results.find(result => result.client === "claude-desktop"); + expect(outcome?.ok).toBe(false); + expect(outcome?.reason).toContain("desired profile changed during sync"); + const persisted = JSON.parse(readFileSync(join(root, "config.json"), "utf8")); + expect(persisted.claudeCode.desktopProfile).toEqual(profileB); + } finally { + refresh.mockRestore(); + aside.mockRestore(); + } + }); + + test("an unchanged desired profile still stores the written fingerprint", async () => { + const profileA = { + version: 1 as const, + assignments: { "mock/hidden": { family: "opus" as const, alias: "claude-opus-4-8-20260201" } }, + defaults: { opus: "mock/hidden", fable: null, sonnet: null, haiku: null }, + }; + const config: OcxConfig = { + port: 10100, + defaultProvider: "mock", + clientIntegrations: { grok: false }, + providers: { + mock: { adapter: "openai-chat", baseUrl: "https://example.test/v1", models: ["keep", "hidden"] }, + openai: { adapter: "openai-responses", baseUrl: "https://example.test/v1", contextWindow: 400_000 }, + }, + apiKeys: [{ id: "sync-key", name: "fixture", key: "ocx_old_sync_fixture", createdAt: "2026-01-01T00:00:00.000Z" }], + claudeCode: { desktopProfile: profileA }, + }; + writeFileSync(join(root, "config.json"), JSON.stringify(config)); + const models: CatalogModel[] = [ + { provider: "mock", id: "keep", contextWindow: 123_000 }, + { provider: "mock", id: "hidden", contextWindow: 456_000 }, + ]; + const realRefresh = ownedRefresh.refreshOwnedIntegration; + const refresh = spyOn(ownedRefresh, "refreshOwnedIntegration").mockImplementation((input, options) => + input.clientId === "mcode" + ? Promise.resolve({ client: "mcode", ok: true, changed: true }) + : realRefresh(input, options)); + const aside = spyOn(asideProfiles, "refreshAsideProfiles").mockResolvedValue([]); + try { + const results = await syncEnabledClientIntegrations(12345, config, { + fetchAllModels: async () => models, + writeDesktop3pConfig: () => ({ written: true, path: "fixture", fingerprint: "0123456789abcdef" }), + }); + expect(results.find(result => result.client === "claude-desktop")).toEqual({ client: "claude-desktop", ok: true, changed: true }); + const persisted = JSON.parse(readFileSync(join(root, "config.json"), "utf8")); + expect(persisted.claudeCode.desktopProfile.appliedFingerprint).toBe("0123456789abcdef"); + expect(persisted.claudeCode.desktopProfile.assignments).toEqual(profileA.assignments); + } finally { + refresh.mockRestore(); + aside.mockRestore(); + } + }); + + const runDesktopSyncWithDrift = async ( + config: OcxConfig, + drift: ((persisted: OcxConfig) => void) | null, + ) => { + writeFileSync(join(root, "config.json"), JSON.stringify(config)); + const models: CatalogModel[] = [ + { provider: "mock", id: "keep", contextWindow: 123_000 }, + { provider: "mock", id: "hidden", contextWindow: 456_000 }, + ]; + const realRefresh = ownedRefresh.refreshOwnedIntegration; + const refresh = spyOn(ownedRefresh, "refreshOwnedIntegration").mockImplementation((input, options) => + input.clientId === "mcode" + ? Promise.resolve({ client: "mcode", ok: true, changed: true }) + : realRefresh(input, options)); + const aside = spyOn(asideProfiles, "refreshAsideProfiles").mockResolvedValue([]); + try { + const results = await syncEnabledClientIntegrations(12345, config, { + fetchAllModels: async () => models, + writeDesktop3pConfig: () => { + // A concurrent writer persists its own desired state while the Desktop + // write is in flight; the marker commit must not overwrite it. + if (drift) { + const drifted = structuredClone(config); + drift(drifted); + writeFileSync(join(root, "config.json"), JSON.stringify(drifted)); + } + return { written: true, path: "fixture", fingerprint: "0123456789abcdef" }; + }, + }); + return { + outcome: results.find(result => result.client === "claude-desktop"), + persisted: JSON.parse(readFileSync(join(root, "config.json"), "utf8")) as OcxConfig, + }; + } finally { + refresh.mockRestore(); + aside.mockRestore(); + } + }; + + const driftProfileA = { + version: 1 as const, + assignments: { "mock/hidden": { family: "opus" as const, alias: "claude-opus-4-8-20260201" } }, + defaults: { opus: "mock/hidden", fable: null, sonnet: null, haiku: null }, + }; + const driftBaseConfig = (claudeCode: OcxConfig["claudeCode"]): OcxConfig => ({ + port: 10100, + defaultProvider: "mock", + clientIntegrations: { grok: false }, + providers: { + mock: { adapter: "openai-chat", baseUrl: "https://example.test/v1", models: ["keep", "hidden"] }, + openai: { adapter: "openai-responses", baseUrl: "https://example.test/v1", contextWindow: 400_000 }, + }, + apiKeys: [{ id: "sync-key", name: "fixture", key: "ocx_old_sync_fixture", createdAt: "2026-01-01T00:00:00.000Z" }], + claudeCode, + }); + + test.each([ + { + name: "a deleted desired profile", + claudeCode: { desktopProfile: driftProfileA, systemEnv: false } as OcxConfig["claudeCode"], + drift: (persisted: OcxConfig) => { delete persisted.claudeCode!.desktopProfile; }, + expectPersisted: (persisted: OcxConfig) => { + expect(persisted.claudeCode).toEqual({ systemEnv: false }); + }, + }, + { + name: "a deleted claudeCode subtree", + claudeCode: { desktopProfile: driftProfileA } as OcxConfig["claudeCode"], + drift: (persisted: OcxConfig) => { delete persisted.claudeCode; }, + expectPersisted: (persisted: OcxConfig) => { + expect(persisted.claudeCode).toBeUndefined(); + }, + }, + { + name: "a deleted explicit empty profile", + claudeCode: { + desktopProfile: { + version: 1 as const, + assignments: {}, + defaults: { opus: null, fable: null, sonnet: null, haiku: null }, + }, + } as OcxConfig["claudeCode"], + drift: (persisted: OcxConfig) => { delete persisted.claudeCode!.desktopProfile; }, + expectPersisted: (persisted: OcxConfig) => { + expect(persisted.claudeCode?.desktopProfile).toBeUndefined(); + }, + }, + ])("sync does not resurrect $name removed during the Desktop write", async ({ claudeCode, drift, expectPersisted }) => { + const { outcome, persisted } = await runDesktopSyncWithDrift(driftBaseConfig(claudeCode), drift); + expect(outcome?.ok).toBe(false); + expect(outcome?.reason).toContain("desired profile changed during sync"); + expectPersisted(persisted); + }); + + test("an initially absent desired profile still stores the written fingerprint", async () => { + const { outcome, persisted } = await runDesktopSyncWithDrift( + driftBaseConfig({ systemEnv: false }), + null, + ); + expect(outcome).toEqual({ client: "claude-desktop", ok: true, changed: true }); + expect(persisted.claudeCode?.desktopProfile?.appliedFingerprint).toBe("0123456789abcdef"); + expect(persisted.claudeCode?.desktopProfile?.assignments).toEqual({}); + expect(persisted.claudeCode?.systemEnv).toBe(false); + }); + + test("sync accepts unchanged defaults with reordered keys during the Desktop write", async () => { + const { outcome, persisted } = await runDesktopSyncWithDrift( + driftBaseConfig({ desktopProfile: driftProfileA }), + config => { + const defaults = config.claudeCode!.desktopProfile!.defaults; + config.claudeCode!.desktopProfile!.defaults = { + haiku: defaults.haiku, + sonnet: defaults.sonnet, + fable: defaults.fable, + opus: defaults.opus, + }; + }, + ); + expect(outcome).toEqual({ client: "claude-desktop", ok: true, changed: true }); + expect(persisted.claudeCode?.desktopProfile?.appliedFingerprint).toBe("0123456789abcdef"); + expect(persisted.claudeCode?.desktopProfile?.defaults).toEqual(driftProfileA.defaults); + }); + }); describe("ocx sync refreshes an already-owned MCode integration", () => { From c6ab36cd37df50a52f7bf2b57ad5d6279682eaf7 Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 07:07:28 +0900 Subject: [PATCH 12/24] fix(claude-desktop): commit applied markers only over the observed baseline Both Desktop writers, provider-change auto-apply and client sync, now capture the desired profile and its applied marker before the Desktop write and commit the new marker only if profile presence, content, fingerprint and timestamp are unchanged. A concurrent edit, deletion or newer marker keeps its state and the write reports a skipped marker. The provider-change path no longer saves a whole stale config snapshot. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- src/claude/desktop-applied-marker.ts | 44 +++++++++++++ .../management/agent-settings-routes.ts | 11 +++- src/server/management/config-routes.ts | 34 ++-------- structure/clients/claude-desktop.md | 15 ++--- .../clients/sync-client-integrations.test.ts | 30 ++++++++- .../native-claude-desktop-toggle.test.ts | 63 +++++++++++++++++++ 6 files changed, 155 insertions(+), 42 deletions(-) create mode 100644 src/claude/desktop-applied-marker.ts diff --git a/src/claude/desktop-applied-marker.ts b/src/claude/desktop-applied-marker.ts new file mode 100644 index 00000000000..e400b30519a --- /dev/null +++ b/src/claude/desktop-applied-marker.ts @@ -0,0 +1,44 @@ +import { mutatePersistedConfig, type PersistedConfigMutationOutcome } from "../config"; +import { emptyDesktopProfile, sameProfileContent, type DesktopProfile } from "./desktop-profile"; + +export interface DesktopAppliedMarkerBaseline { + profile: DesktopProfile | undefined; + appliedFingerprint: string | undefined; + appliedAt: string | undefined; +} + +export function captureDesktopAppliedMarker( + profile: DesktopProfile | undefined, +): DesktopAppliedMarkerBaseline { + const snapshot = profile == null ? undefined : structuredClone(profile); + return { + profile: snapshot, + appliedFingerprint: snapshot?.appliedFingerprint, + appliedAt: snapshot?.appliedAt, + }; +} + +export function commitDesktopAppliedMarker( + baseline: DesktopAppliedMarkerBaseline, + fingerprint: string, +): PersistedConfigMutationOutcome { + const appliedAt = new Date().toISOString(); + return mutatePersistedConfig(persisted => { + const profile = persisted.claudeCode?.desktopProfile; + if ((profile == null) !== (baseline.profile === undefined) + || (profile && baseline.profile && !sameProfileContent(profile, baseline.profile)) + || profile?.appliedFingerprint !== baseline.appliedFingerprint + || profile?.appliedAt !== baseline.appliedAt) { + return { changed: false, value: false }; + } + persisted.claudeCode = { + ...(persisted.claudeCode ?? {}), + desktopProfile: { + ...(profile ?? emptyDesktopProfile()), + appliedFingerprint: fingerprint, + appliedAt, + }, + }; + return { changed: true, value: true }; + }); +} diff --git a/src/server/management/agent-settings-routes.ts b/src/server/management/agent-settings-routes.ts index 845a754474c..77c1319d334 100644 --- a/src/server/management/agent-settings-routes.ts +++ b/src/server/management/agent-settings-routes.ts @@ -1,4 +1,5 @@ import { persistCommittedDesktopGateway } from "../../claude/desktop-gateway-state"; +import { captureDesktopAppliedMarker, commitDesktopAppliedMarker } from "../../claude/desktop-applied-marker"; import { randomUUID } from "node:crypto"; import { readFileSync } from "node:fs"; import type { CatalogModel } from "../../codex/catalog"; @@ -231,18 +232,22 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise }).kind; if (["not_installed", "no_owned_state", "foreign", "unsafe", "broken"].includes(afterKind)) return; const routed = filterCatalogVisibleModels(allModels, current).map(m => ({ provider: m.provider, id: m.id, contextWindow: m.contextWindow })); + const writtenProfile = current.claudeCode.desktopProfile; + const markerBaseline = captureDesktopAppliedMarker(writtenProfile); const result = (deps.writeDesktop3pConfig ?? writeDesktop3pConfig)( current.port ?? 10100, [...desktopVisibleNativeSlugs(current)], routed, current.apiKeys?.[0]?.key, "static", - current.claudeCode.desktopProfile, + writtenProfile, nativeContextLimits(current), ); if (result.written && result.fingerprint) { - current.claudeCode = { ...current.claudeCode, desktopProfile: { ...current.claudeCode.desktopProfile, appliedFingerprint: result.fingerprint, appliedAt: new Date().toISOString() } }; - saveConfigPreservingClaudeCode(current); + const marked = commitDesktopAppliedMarker(markerBaseline, result.fingerprint); + if (marked.status === "unavailable" || marked.value === false) { + console.warn("[claude-desktop] provider-change applied marker skipped"); + } } } catch { /* best-effort */ } } diff --git a/src/server/management/config-routes.ts b/src/server/management/config-routes.ts index 77e7dec61f0..6bf49819aab 100644 --- a/src/server/management/config-routes.ts +++ b/src/server/management/config-routes.ts @@ -18,11 +18,11 @@ import { isValidProviderName, loadConfig, multiAgentGuidanceEnabled, - mutatePersistedConfig, providerBaseUrlConfigError, providerHeadersConfigError, saveConfigPreservingClaudeCode, } from "../../config"; +import { captureDesktopAppliedMarker, commitDesktopAppliedMarker } from "../../claude/desktop-applied-marker"; import { clearLoginState, getLoginStatus, @@ -220,45 +220,21 @@ export async function syncEnabledClientIntegrations( if (claudeDesktopIntegrationEnabled(latest)) { const routed = filterCatalogVisibleModels(models, latest) .map(model => ({ provider: model.provider, id: model.id, contextWindow: model.contextWindow })); + const writtenProfile = latest.claudeCode?.desktopProfile; + const markerBaseline = captureDesktopAppliedMarker(writtenProfile); const r = (deps.writeDesktop3pConfig ?? writeDesktop3pConfig)( port, [...desktopVisibleNativeSlugs(latest)], routed, latest.apiKeys?.[0]?.key, "static", - latest.claudeCode?.desktopProfile, + writtenProfile, nativeContextLimits(latest), ); if (!r.written || !r.fingerprint) { out.push({ client: "claude-desktop", ok: false, reason: r.reason ?? "Claude Desktop write failed" }); } else { - const { emptyDesktopProfile, sameProfileContent } = await import("../../claude/desktop-profile"); - // The fingerprint belongs to the desired profile the write just used. If another - // writer saved a different desired profile between the Desktop write and this marker - // commit, stamping it would claim B is applied while the disk holds A's bytes. - const writtenProfile = latest.claudeCode?.desktopProfile; - const marked = mutatePersistedConfig(persisted => { - const profile = persisted.claudeCode?.desktopProfile; - // Presence first: a concurrent delete (of the profile or the whole claudeCode - // subtree) must not resurrect the written profile under a fresh fingerprint, - // and a concurrent insert must not inherit it either. Content is compared only - // when both sides carry a profile. - if ((profile == null) !== (writtenProfile == null)) { - return { changed: false, value: false }; - } - if (profile && writtenProfile && !sameProfileContent(profile, writtenProfile)) { - return { changed: false, value: false }; - } - persisted.claudeCode = { - ...(persisted.claudeCode ?? {}), - desktopProfile: { - ...(profile ?? writtenProfile ?? emptyDesktopProfile()), - appliedFingerprint: r.fingerprint, - appliedAt: new Date().toISOString(), - }, - }; - return { changed: true, value: true }; - }); + const marked = commitDesktopAppliedMarker(markerBaseline, r.fingerprint); out.push(marked.status === "unavailable" ? { client: "claude-desktop", ok: false, reason: "Claude Desktop applied marker was not saved (" + marked.reason + ")" } : marked.value === false diff --git a/structure/clients/claude-desktop.md b/structure/clients/claude-desktop.md index 656855e6002..ef82cb9ab2a 100644 --- a/structure/clients/claude-desktop.md +++ b/structure/clients/claude-desktop.md @@ -152,13 +152,14 @@ not clear a newer connection. Authorized uninstall completes or resumes owned De before removing OpenCodex state, and preserves recovery state when cleanup conflicts or fails. The server-owned applied marker (`claudeCode.desktopProfile.appliedFingerprint` and -`appliedAt`) is committed by the config route only while the persisted desired profile still -matches the profile that was just written. Presence is compared first, then content: if a concurrent -writer deleted the profile or the whole `claudeCode` block, or replaced it with a different -profile, before the marker commit, the write is declined and reported as skipped instead of -resurrecting the removed profile with a fresh fingerprint. A profile that was absent from the start -still stores its fingerprint normally. Default-family key order does not change the desired content; -the comparison uses each family's selected route while preserving real selection changes. +`appliedAt`) is committed through `src/claude/desktop-applied-marker.ts` only while the +persisted desired profile still matches the exact profile handed to the Desktop writer and +its prior fingerprint and time are unchanged. Sync compares profile presence, content and +both marker fields before committing; an initially absent profile can receive a marker, while +a concurrently deleted or changed profile or a newer marker is left intact and the existing +skip outcome is reported. Provider-change auto-apply requires a present profile and emits a +generic diagnostic when the same comparison declines its marker. Default-family key order +does not change desired content; the comparison uses each family's selected route. These guarantees concern files on disk. Fully quitting and reopening Desktop is required after apply, rotation/recovery or restoration; there is no automatic process restart or guarantee that diff --git a/tests/clients/sync-client-integrations.test.ts b/tests/clients/sync-client-integrations.test.ts index 3f6b4b4bcb2..874d760f052 100644 --- a/tests/clients/sync-client-integrations.test.ts +++ b/tests/clients/sync-client-integrations.test.ts @@ -72,9 +72,10 @@ describe("ocx sync fans out to enabled native clients and owned file integration expect(fn).toContain("nativeContextLimits(latest)"); // Cleanup accepts only the fingerprint of the exact credential-bearing profile we wrote. // Sync must durably advance that ownership marker rather than leaving the old value behind. - expect(fn).toContain("mutatePersistedConfig(persisted =>"); - expect(fn).toContain("appliedFingerprint: r.fingerprint"); - expect(fn.indexOf("nativeContextLimits(latest)")).toBeLessThan(fn.indexOf("appliedFingerprint: r.fingerprint")); + expect(fn).toContain("captureDesktopAppliedMarker(writtenProfile)"); + expect(fn).toContain("commitDesktopAppliedMarker(markerBaseline, r.fingerprint)"); + expect(fn.indexOf("captureDesktopAppliedMarker(writtenProfile)")).toBeLessThan(fn.indexOf("const r = (deps.writeDesktop3pConfig")); + expect(fn.indexOf("nativeContextLimits(latest)")).toBeLessThan(fn.indexOf("commitDesktopAppliedMarker(markerBaseline, r.fingerprint)")); // A client that is off is omitted rather than reported: the caller has to be able to // tell "left alone" from "tried and failed", so there is no skipped state to emit. expect(fn).not.toContain('"skipped"'); @@ -362,6 +363,29 @@ describe("Desktop sync rechecks persisted state after discovery", () => { claudeCode, }); + test.each([ + { name: "newer fingerprint", fingerprint: "newer-fingerprint" }, + { name: "newer timestamp only", fingerprint: "prior-fingerprint" }, + ])("sync preserves a $name committed during the Desktop write", async ({ fingerprint }) => { + const initialProfile = { + ...driftProfileA, + appliedFingerprint: "prior-fingerprint", + appliedAt: "2026-09-23T00:00:00.000Z", + }; + const newerProfile = { + ...initialProfile, + appliedFingerprint: fingerprint, + appliedAt: "2026-09-23T00:00:01.000Z", + }; + const { outcome, persisted } = await runDesktopSyncWithDrift( + driftBaseConfig({ desktopProfile: initialProfile }), + concurrent => { concurrent.claudeCode!.desktopProfile = newerProfile; }, + ); + expect(outcome?.ok).toBe(false); + expect(outcome?.reason).toContain("applied marker skipped"); + expect(persisted.claudeCode?.desktopProfile).toEqual(newerProfile); + }); + test.each([ { name: "a deleted desired profile", diff --git a/tests/codex-integration/native-claude-desktop-toggle.test.ts b/tests/codex-integration/native-claude-desktop-toggle.test.ts index ffe20377f0d..084cc95143f 100644 --- a/tests/codex-integration/native-claude-desktop-toggle.test.ts +++ b/tests/codex-integration/native-claude-desktop-toggle.test.ts @@ -285,6 +285,69 @@ test("auto-apply re-reads desired state after catalog fetch and skips a concurre expect(writes).toBe(0); }); +test("provider-change auto-apply preserves concurrent Desktop profile edits, deletions, and newer markers", async () => { + const profileA = { + version: 1 as const, + assignments: {}, + defaults: { opus: null, fable: null, sonnet: null, haiku: null }, + appliedFingerprint: "prior-fingerprint", + appliedAt: "2026-09-23T00:00:00.000Z", + }; + const profileB = { + version: 1 as const, + assignments: { "mock/test-model": { family: "sonnet" as const, alias: "claude-opus-4-8-20260202" } }, + defaults: { opus: null, fable: null, sonnet: "mock/test-model", haiku: null }, + }; + const id = "selected-owned"; + mkdirSync(library); + writeFileSync(join(library, "_meta.json"), JSON.stringify({ appliedId: id, entries: [{ id, name: "opencodex" }] })); + writeFileSync(join(library, `${id}.json`), JSON.stringify({ + inferenceProvider: "gateway", inferenceCredentialKind: "static", + inferenceGatewayBaseUrl: "fixture", inferenceGatewayApiKey: "not-a-secret", + })); + + for (const change of ["edit", "delete-profile", "delete-subtree", "newer-marker", "newer-time"] as const) { + const starting = { ...config(), claudeCode: { desktopMode: "gateway" as const, desktopProfile: profileA, injectAgents: false } }; + writeFileSync(join(root, "config.json"), JSON.stringify(starting)); + let writes = 0; + const response = await dispatch("/api/subagent-models", { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ models: [] }), + }, { + fetchAllModels: async () => [], + writeDesktop3pConfig: (_port, _slugs, _models, _key, _mode, profile) => { + writes++; + expect(profile).toEqual(profileA); + const newer = JSON.parse(readFileSync(join(root, "config.json"), "utf8")) as OcxConfig; + if (change === "edit") newer.claudeCode = { ...newer.claudeCode, desktopProfile: profileB }; + if (change === "delete-profile") delete newer.claudeCode!.desktopProfile; + if (change === "delete-subtree") delete newer.claudeCode; + if (change === "newer-marker") newer.claudeCode!.desktopProfile = { + ...profileA, appliedFingerprint: "newer-fingerprint", appliedAt: "2026-09-23T00:00:01.000Z", + }; + if (change === "newer-time") newer.claudeCode!.desktopProfile = { + ...profileA, appliedAt: "2026-09-23T00:00:01.000Z", + }; + writeFileSync(join(root, "config.json"), JSON.stringify(newer)); + return { written: true, path: join(library, "new.json"), fingerprint: "0123456789abcdef" }; + }, + }, starting); + expect(response?.status).toBe(200); + expect(writes).toBe(1); + const saved = JSON.parse(readFileSync(join(root, "config.json"), "utf8")) as OcxConfig; + if (change === "edit") expect(saved.claudeCode?.desktopProfile).toEqual(profileB); + if (change === "delete-profile") expect(saved.claudeCode?.desktopProfile).toBeUndefined(); + if (change === "delete-subtree") expect(saved.claudeCode).toBeUndefined(); + if (change === "newer-marker") expect(saved.claudeCode?.desktopProfile).toEqual({ + ...profileA, appliedFingerprint: "newer-fingerprint", appliedAt: "2026-09-23T00:00:01.000Z", + }); + if (change === "newer-time") expect(saved.claudeCode?.desktopProfile).toEqual({ + ...profileA, appliedAt: "2026-09-23T00:00:01.000Z", + }); + } +}); + test("explicit enable re-reads desired state after catalog fetch and skips a concurrent OFF", async () => { let release!: () => void; let started!: () => void; From bda6c07105571ac127dc2cda1bf3b9278e38c118 Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 07:21:52 +0900 Subject: [PATCH 13/24] fix(claude-desktop): commit profile edits against the persisted marker The Desktop profile PUT built its response from an earlier snapshot and saved that whole snapshot, so a marker committed by another writer during the awaited state build could be replaced by an older one. The edit now commits in one persisted-config mutation that keeps the latest marker for unchanged content and answers 409 when the profile itself changed meanwhile. The Meta Muse overflow test now asserts that the bounded-body limit, not a generic failure, produced the error. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- .../management/agent-settings-routes.ts | 37 ++++++++--- structure/clients/claude-desktop.md | 6 ++ .../claude-management-api.test.ts | 61 ++++++++++++++++++- tests/oauth/oauth-public-surface.test.ts | 21 +++++-- 4 files changed, 111 insertions(+), 14 deletions(-) diff --git a/src/server/management/agent-settings-routes.ts b/src/server/management/agent-settings-routes.ts index 77c1319d334..35213b4fc3b 100644 --- a/src/server/management/agent-settings-routes.ts +++ b/src/server/management/agent-settings-routes.ts @@ -990,9 +990,10 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise let body: { profile?: unknown }; try { body = await readManagementJsonBody(req); } catch (error) { rethrowManagementBodyTooLarge(error); return jsonResponse({ error: "invalid JSON body" }, 400); } try { - const { parseDesktopProfile, preserveDesktopAppliedState, reconcileDesktopProfile } = await import("../../claude/desktop-profile"); + const { parseDesktopProfile, preserveDesktopAppliedState, reconcileDesktopProfile, sameProfileContent } = await import("../../claude/desktop-profile"); const parsed = parseDesktopProfile(body.profile); - const current = await buildClaudeDesktopState(config); + const initial = loadConfig(); + const current = await buildClaudeDesktopState(initial); const availableRoutes = new Set(current.models.filter(item => item.available).map(item => item.route)); for (const route of Object.keys(parsed.assignments)) { if (!current.profile.assignments[route] && !availableRoutes.has(route)) { @@ -1016,13 +1017,33 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise // Applied markers are server-owned bookkeeping. Discard client copies, then // restore the trusted markers only if the desired profile stayed identical. const editable = { version: 1 as const, assignments: parsed.assignments, defaults: parsed.defaults }; - const state = await buildClaudeDesktopState(config, editable); + const state = await buildClaudeDesktopState(initial, editable); const rebuilt = reconcileDesktopProfile(state.profile, state.models); - config.claudeCode = { - ...(config.claudeCode ?? {}), - desktopProfile: preserveDesktopAppliedState(current.profile, rebuilt), - }; - saveConfigPreservingClaudeCode(config); + const expectedProfile = initial.claudeCode?.desktopProfile; + const outcome = mutatePersistedConfig(persisted => { + const latest = persisted.claudeCode?.desktopProfile; + if ((latest == null) !== (expectedProfile == null) + || (latest && expectedProfile && !sameProfileContent(latest, expectedProfile))) { + return { changed: false, value: null }; + } + persisted.claudeCode = { + ...(persisted.claudeCode ?? {}), + desktopProfile: preserveDesktopAppliedState(latest ?? current.profile, rebuilt), + }; + return { changed: true, value: structuredClone(persisted.claudeCode) }; + }); + if (outcome.status === "unavailable" || outcome.value === null) { + return jsonResponse({ error: "Claude Desktop profile changed during save" }, 409); + } + adoptPersistedClaudeCode(config, outcome.value); + // An unarmed live snapshot may prefer its old marker during reconciliation. + // Pin the field this route just committed while leaving other live leaves intact. + if (outcome.value?.desktopProfile) { + config.claudeCode = { + ...(config.claudeCode ?? {}), + desktopProfile: structuredClone(outcome.value.desktopProfile), + }; + } const saved = await buildClaudeDesktopState(config); const runtimePort = Number(url.port) || config.port; return jsonResponse({ ok: true, ...saved, port: runtimePort }); diff --git a/structure/clients/claude-desktop.md b/structure/clients/claude-desktop.md index ef82cb9ab2a..2e12a2215f5 100644 --- a/structure/clients/claude-desktop.md +++ b/structure/clients/claude-desktop.md @@ -161,6 +161,12 @@ skip outcome is reported. Provider-change auto-apply requires a present profile generic diagnostic when the same comparison declines its marker. Default-family key order does not change desired content; the comparison uses each family's selected route. +The profile PUT in `src/server/management/agent-settings-routes.ts` validates against a +persisted profile snapshot and commits only `claudeCode.desktopProfile` under the config +mutation lock. Client marker fields are discarded. Unchanged desired content keeps the +latest persisted marker, including one committed while the PUT awaited model discovery; +a concurrent desired-profile edit declines the PUT with 409 instead of being overwritten. + These guarantees concern files on disk. Fully quitting and reopening Desktop is required after apply, rotation/recovery or restoration; there is no automatic process restart or guarantee that a running app discarded a key. Local disconnect does not revoke the hub key or remove arbitrary diff --git a/tests/claude-integration/claude-management-api.test.ts b/tests/claude-integration/claude-management-api.test.ts index 8c4a7994b16..b99f865be13 100644 --- a/tests/claude-integration/claude-management-api.test.ts +++ b/tests/claude-integration/claude-management-api.test.ts @@ -3,9 +3,12 @@ import { managementFetch as fetch } from "../helpers/management-auth"; import { mkdtempSync, readdirSync, readFileSync} from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; -import { loadConfig, saveConfig } from "../../src/config"; +import { loadConfig, saveConfig, saveConfigPreservingClaudeCode } from "../../src/config"; import { startServer as startServerImpl } from "../../src/server"; +import { handleManagementAPI } from "../../src/server/management-api"; import { writeDesktop3pConfig, removeDesktop3pStandardPivot } from "../../src/claude/desktop-3p"; +import * as desktopProfiles from "../../src/claude/desktop-profile"; +import { buildClaudeDesktopState } from "../../src/server/management/shared"; import * as systemEnv from "../../src/server/system-env"; import type { OcxConfig } from "../../src/types"; import { installIsolatedCodexHome, type IsolatedCodexHome } from "../helpers/isolated-codex-home"; @@ -907,6 +910,62 @@ test("Claude Desktop PUT clears applied markers when routing changes", async () } }); +test("Claude Desktop PUT preserves a newer applied marker committed during profile rebuilding", async () => { + const seeded = loadConfig(); + const initialState = await buildClaudeDesktopState(seeded); + seeded.claudeCode = { + ...(seeded.claudeCode ?? {}), + desktopProfile: { + ...initialState.profile, + appliedFingerprint: "older-fingerprint", + appliedAt: "2026-09-23T00:00:00.000Z", + }, + }; + saveConfig(seeded); + // This standalone management snapshot has no live-config baseline. The old + // whole-snapshot save would therefore overwrite the newer disk marker. + const liveConfig = structuredClone(seeded); + const originalReconcile = desktopProfiles.reconcileDesktopProfile; + let builds = 0; + let injected = false; + const reconcile = spyOn(desktopProfiles, "reconcileDesktopProfile").mockImplementation((stored, models) => { + const profile = originalReconcile(stored, models); + builds += 1; + if (builds === 2) { + const concurrent = loadConfig(); + concurrent.claudeCode = { + ...(concurrent.claudeCode ?? {}), + desktopProfile: { + ...concurrent.claudeCode!.desktopProfile!, + appliedFingerprint: "newer-fingerprint", + appliedAt: "2026-09-23T00:00:01.000Z", + }, + }; + saveConfig(concurrent); + injected = true; + } + return profile; + }); + try { + const url = new URL("http://127.0.0.1:10100/api/claude-desktop"); + const req = new Request(url, { + method: "PUT", + headers: { Host: url.host, "Content-Type": "application/json" }, + body: JSON.stringify({ profile: initialState.profile }), + }); + const put = await handleManagementAPI(req, url, liveConfig); + expect(put?.status).toBe(200); + expect(injected).toBe(true); + expect(loadConfig().claudeCode?.desktopProfile?.appliedFingerprint).toBe("newer-fingerprint"); + expect(loadConfig().claudeCode?.desktopProfile?.appliedAt).toBe("2026-09-23T00:00:01.000Z"); + expect(liveConfig.claudeCode?.desktopProfile?.appliedFingerprint).toBe("newer-fingerprint"); + saveConfigPreservingClaudeCode(liveConfig); + expect(loadConfig().claudeCode?.desktopProfile?.appliedFingerprint).toBe("newer-fingerprint"); + } finally { + reconcile.mockRestore(); + } +}); + test("Claude Desktop PUT retains but cannot move an unavailable route", async () => { const seeded = loadConfig(); seeded.claudeCode = { diff --git a/tests/oauth/oauth-public-surface.test.ts b/tests/oauth/oauth-public-surface.test.ts index 4937d4cddbc..5051d6b272e 100644 --- a/tests/oauth/oauth-public-surface.test.ts +++ b/tests/oauth/oauth-public-surface.test.ts @@ -19,7 +19,7 @@ import type { OAuthController } from "../../src/oauth/types"; import { getCredential } from "../../src/oauth/store"; import * as oauthStore from "../../src/oauth/store"; import * as oauth from "../../src/oauth"; -import { requestMuseDeviceAuthorization } from "../../src/oauth/meta-muse-device"; +import { MuseDeviceLoginError, requestMuseDeviceAuthorization } from "../../src/oauth/meta-muse-device"; import { flushConfigDirHardeningForTests } from "../../src/config/paths"; import { setAsyncIcaclsRunnerForTests, setIcaclsRunnerForTests } from "../../src/lib/windows-secret-acl"; @@ -150,11 +150,21 @@ describe("legacy ChatGPT OAuth public-surface exclusion", () => { const cfg = config(); saveConfig(cfg); let fetches = 0; + let overflowObserved = false; const login = spyOn(oauth, "startLoginFlow").mockImplementation(async () => { - await requestMuseDeviceAuthorization({ fetchImpl: (async () => { - fetches++; - return new Response(JSON.stringify({ device_code: "private-device-canary", filler: "x".repeat(65_536) })); - }) as typeof fetch }); + try { + await requestMuseDeviceAuthorization({ fetchImpl: (async () => { + fetches++; + return new Response(JSON.stringify({ device_code: "private-device-canary", filler: "x".repeat(65_536) })); + }) as typeof fetch }); + } catch (error) { + expect(error).toBeInstanceOf(MuseDeviceLoginError); + if (!(error instanceof MuseDeviceLoginError)) throw error; + expect(error.kind).toBe("device-authorization"); + expect(error.message).toContain("exceeded the 65536-byte limit"); + overflowObserved = true; + throw error; + } throw new Error("oversized authorization must not succeed"); }); const request = () => new Request("http://localhost/api/oauth/login", { @@ -170,6 +180,7 @@ describe("legacy ChatGPT OAuth public-surface exclusion", () => { expect(response?.status).toBe(409); expect(await response?.json()).toEqual({ error: PUBLIC_OAUTH_ERROR }); expect(fetches).toBe(1); + expect(overflowObserved).toBe(true); expect(getCredential("meta-muse")).toBeNull(); } finally { login.mockRestore(); } }); From 4a521eda570b1c944f026c4c565d986e8151049f Mon Sep 17 00:00:00 2001 From: JUN Date: Wed, 23 Sep 2026 07:25:20 +0900 Subject: [PATCH 14/24] fix(claude-desktop): report an unreadable config separately from an edit conflict A missing or invalid config now answers 500 with its reason; only a concurrent profile change or exhausted rebase answers 409. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- src/server/management/agent-settings-routes.ts | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/server/management/agent-settings-routes.ts b/src/server/management/agent-settings-routes.ts index 35213b4fc3b..7c5f6af3605 100644 --- a/src/server/management/agent-settings-routes.ts +++ b/src/server/management/agent-settings-routes.ts @@ -1032,6 +1032,9 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise }; return { changed: true, value: structuredClone(persisted.claudeCode) }; }); + if (outcome.status === "unavailable" && outcome.reason !== "conflict") { + return jsonResponse({ error: "Claude Desktop profile could not be saved (config " + outcome.reason + ")" }, 500); + } if (outcome.status === "unavailable" || outcome.value === null) { return jsonResponse({ error: "Claude Desktop profile changed during save" }, 409); } From bf1a64ac298bc81af4e04162256b9a610806df47 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Wed, 23 Sep 2026 08:15:17 +0900 Subject: [PATCH 15/24] feat(desktop): consolidate consent-based runtime takeover and ownership contracts (#5564) Carries #5564, which consolidates #5459 and #5457, onto current dev. The review screenshot stays in the pull request description rather than the tree. Co-authored-by: jun Co-authored-by: sanggyulee Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- desktop/src-tauri/src/claim.rs | 289 ++++++ desktop/src-tauri/src/lib.rs | 15 +- desktop/src-tauri/src/ownership.rs | 134 ++- desktop/src-tauri/src/resolve.rs | 104 ++- desktop/src-tauri/src/startup.rs | 856 ++++++++++++++++-- desktop/ui/index.html | 50 +- src/cli/registry.ts | 2 +- src/cli/resolve.ts | 96 +- src/service/claim.ts | 185 ++++ src/service/cli.ts | 11 +- src/service/managing-cli.ts | 158 ++++ structure/desktop-shell.md | 41 +- tests/cli/cli-help.test.ts | 2 +- tests/cli/cli-resolve.test.ts | 155 +++- tests/clients/desktop-cli-contracts.test.ts | 3 +- .../clients/desktop-install-identity.test.ts | 81 +- .../clients/desktop-runtime-identity.test.ts | 2 +- tests/clients/desktop-startup-surface.test.ts | 33 +- tests/service/service-claim.test.ts | 155 ++++ 19 files changed, 2200 insertions(+), 172 deletions(-) create mode 100644 desktop/src-tauri/src/claim.rs create mode 100644 src/service/claim.ts create mode 100644 src/service/managing-cli.ts create mode 100644 tests/service/service-claim.test.ts diff --git a/desktop/src-tauri/src/claim.rs b/desktop/src-tauri/src/claim.rs new file mode 100644 index 00000000000..22d0cfcbc24 --- /dev/null +++ b/desktop/src-tauri/src/claim.rs @@ -0,0 +1,289 @@ +//! Recording this installation as the runtime owner, through the bundled CLI. +//! +//! The claim is only valid against the exact answer the consent prompt was approved from, +//! so this is a subprocess with expectations on argv rather than an in-process write: the +//! ownership mutation lease, the subject revalidation and the managing-CLI re-observation +//! all live in the CLI's `recordServiceOwner`, and re-running them here would be a second +//! implementation of a rule that has to be identical. +//! +//! Like `runtime_stop`, the result is a document, not a guess: `ocx service claim --json` +//! puts one summary on stdout and this consumes `ok` and the exit code rather than +//! inferring them. A claim that did not end in exit 0 with `ok:true` is a claim that did +//! not happen — and a takeover that reached here already stopped the foreign runtime, so +//! the caller's failure is a stopped runtime with no owner recorded, which the next launch +//! resolves as an ordinary absence. + +use serde::Deserialize; +use tauri::AppHandle; +use tauri_plugin_shell::ShellExt; +use tokio::time::{timeout_at, Instant}; + +/// The wire version this shell understands. +pub const SCHEMA: &str = "ocx-service-claim/1"; + +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ClaimOwnership { + pub owner: String, + pub install_id: String, + pub consent_generation: u64, +} + +#[derive(Clone, Debug, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ClaimSummary { + pub schema: String, + pub ok: bool, + /// Present on success. + pub ownership: Option, + /// Present on failure: the CLI's machine-readable error code. + pub code: Option, + /// Present on failure. + pub message: Option, +} + +/// What the shell concluded. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum ClaimResult { + /// The CLI recorded the claim and named the generation it landed at. + Recorded(ClaimOwnership), + /// It reported anything else, or the run could not be read at all. + Failed(String), +} + +impl ClaimResult { + #[cfg(test)] + pub fn is_recorded(&self) -> bool { + matches!(self, Self::Recorded(_)) + } +} + +/// The arguments a takeover builds from the resolve answer it was approved against. +/// +/// `Recorded::Unknown` gets no argv: fabricating `--expect-none --expect-revision 0` would claim +/// against a subject nobody approved, so the answer is None and the caller refuses. +pub fn args( + install_id: &str, + recorded: &crate::ownership::Recorded, + token: &str, +) -> Option> { + let mut argv = vec![ + "service".to_owned(), + "claim".to_owned(), + "--owner".to_owned(), + "desktop".to_owned(), + "--install-id".to_owned(), + install_id.to_owned(), + ]; + match recorded { + crate::ownership::Recorded::None { revision } => { + argv.push("--expect-none".to_owned()); + argv.push("--expect-revision".to_owned()); + argv.push(revision.to_string()); + } + crate::ownership::Recorded::Owned { + ownership, + revision, + } => { + argv.extend([ + "--expect-owner".to_owned(), + match ownership.owner { + crate::ownership::Owner::Cli => "cli".to_owned(), + crate::ownership::Owner::Desktop => "desktop".to_owned(), + }, + "--expect-install-id".to_owned(), + ownership.install_id.clone(), + "--expect-generation".to_owned(), + ownership.consent_generation.to_string(), + "--expect-revision".to_owned(), + revision.to_string(), + ]); + } + // A takeover is only offered when the record was read; unknown never reaches here, + // and refusing beats inventing an approval. + crate::ownership::Recorded::Unknown { .. } => return None, + } + argv.extend([ + "--expect-compatibility-token".to_owned(), + token.to_owned(), + "--json".to_owned(), + ]); + Some(argv) +} + +/// Read one claim summary. +/// +/// Exit 0 with `ok:true` is the only success — the claim path uses exit 1 with a +/// machine-readable `code` for subject mismatches and changed compatibility, and both of +/// those are refusals to re-ask from, not partial writes. +pub fn read(exit_code: Option, stdout: &[u8], stderr: &[u8]) -> ClaimResult { + let text = String::from_utf8_lossy(stdout); + let summary: ClaimSummary = match serde_json::from_str(text.trim()) { + Ok(summary) => summary, + Err(error) => { + let detail = String::from_utf8_lossy(stderr); + let detail = detail.trim(); + let code = exit_code + .map(|code| code.to_string()) + .unwrap_or_else(|| "no exit code".to_owned()); + return ClaimResult::Failed(if detail.is_empty() { + format!("the bundled CLI's claim output could not be read (exit {code}: {error})") + } else { + format!("the bundled CLI's claim output could not be read (exit {code}): {detail}") + }); + } + }; + if summary.schema != SCHEMA { + return ClaimResult::Failed(format!( + "the bundled CLI answered with schema {} and this app understands {SCHEMA}", + summary.schema + )); + } + if exit_code != Some(0) || !summary.ok { + return ClaimResult::Failed(summary.message.unwrap_or_else(|| { + format!( + "the claim was refused ({})", + summary.code.unwrap_or_else(|| "no code".to_owned()) + ) + })); + } + match summary.ownership { + Some(ownership) => ClaimResult::Recorded(ownership), + None => ClaimResult::Failed( + "the claim reported success but carried no ownership record".to_owned(), + ), + } +} + +/// Run the bundled `ocx service claim`, under the caller's deadline. +pub async fn run(app: &AppHandle, argv: Vec, deadline: Instant) -> ClaimResult { + let command = match app.shell().sidecar("ocx") { + Ok(command) => command.args(argv), + Err(error) => { + return ClaimResult::Failed(format!("the bundled CLI could not be started ({error})")) + } + }; + match timeout_at(deadline, command.output()).await { + Ok(Ok(output)) => read(output.status.code(), &output.stdout, &output.stderr), + Ok(Err(error)) => { + ClaimResult::Failed(format!("the bundled CLI could not be run ({error})")) + } + Err(_) => ClaimResult::Failed( + "the bundled CLI did not finish the claim before the deadline".to_owned(), + ), + } +} + +#[cfg(test)] +mod tests { + use super::{args, read, ClaimResult}; + use crate::ownership::{Owner, Recorded}; + + fn document(ok: bool, extra: &str) -> String { + format!(r#"{{"schema":"ocx-service-claim/1","ok":{ok}{extra}}}"#) + } + + #[test] + fn the_arguments_carry_the_exact_approved_subject() { + let none = args("install-a", &Recorded::None { revision: 0 }, "tok"); + assert_eq!( + none.expect("argv for a read record"), + vec![ + "service", + "claim", + "--owner", + "desktop", + "--install-id", + "install-a", + "--expect-none", + "--expect-revision", + "0", + "--expect-compatibility-token", + "tok", + "--json", + ] + ); + let owned = Recorded::Owned { + ownership: crate::ownership::Claim { + owner: Owner::Cli, + install_id: "npm-1".to_owned(), + consent_generation: 2, + }, + revision: 9, + }; + let argv = args("install-a", &owned, "tok").expect("a claim against a read record"); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-owner", "cli"])); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-install-id", "npm-1"])); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-generation", "2"])); + assert!(argv + .windows(2) + .any(|pair| pair == ["--expect-revision", "9"])); + } + + #[test] + fn an_unread_record_gets_no_claim_rather_than_a_fabricated_one() { + // Nobody approved a subject the resolve could not read, so there is nothing to claim + // against -- and "expect none, revision 0" would be that approval invented. + assert!(args( + "install-a", + &Recorded::Unknown { + reason: "why".to_owned() + }, + "tok" + ) + .is_none()); + } + + #[test] + fn a_recorded_claim_is_the_only_success() { + let ok = document( + true, + r#","ownership":{"owner":"desktop","installId":"install-a","consentGeneration":1},"revision":3"#, + ); + let result = read(Some(0), ok.as_bytes(), b""); + match result { + ClaimResult::Recorded(ownership) => { + assert_eq!(ownership.install_id, "install-a"); + assert_eq!(ownership.consent_generation, 1); + } + ClaimResult::Failed(reason) => panic!("{reason}"), + } + // Success has to arrive with exit 0 and the record it wrote. + assert!(!read(Some(1), ok.as_bytes(), b"").is_recorded()); + assert!(!read(Some(0), document(true, "").as_bytes(), b"").is_recorded()); + } + + #[test] + fn a_refusal_carries_the_clis_own_message() { + let refused = document( + false, + r#","code":"service-ownership-subject-mismatch","message":"ownership changed""#, + ); + let result = read(Some(1), refused.as_bytes(), b""); + match result { + ClaimResult::Failed(reason) => assert!(reason.contains("ownership changed")), + ClaimResult::Recorded(_) => panic!("a refused claim is not recorded"), + } + } + + #[test] + fn output_that_cannot_be_read_is_a_failure_not_a_claim() { + assert!(!read(Some(0), b"", b"boom").is_recorded()); + assert!(!read(Some(0), b"not json", b"").is_recorded()); + assert!(!read(None, b"", b"").is_recorded()); + let future = document(true, r#","ownership":{"owner":"desktop","installId":"i","consentGeneration":1},"revision":1"#) + .replace("ocx-service-claim/1", "ocx-service-claim/2"); + let result = read(Some(0), future.as_bytes(), b""); + assert!(!result.is_recorded()); + match result { + ClaimResult::Failed(reason) => assert!(reason.contains("ocx-service-claim/2")), + _ => unreachable!(), + } + } +} diff --git a/desktop/src-tauri/src/lib.rs b/desktop/src-tauri/src/lib.rs index 7464a3692c0..e9467013ebb 100644 --- a/desktop/src-tauri/src/lib.rs +++ b/desktop/src-tauri/src/lib.rs @@ -1,4 +1,5 @@ mod auth; +mod claim; #[cfg(target_os = "macos")] mod companion_query; mod companion_usage; @@ -176,6 +177,17 @@ fn retry_startup(app: tauri::AppHandle) { startup::begin(&app); } +/// The user's answer to the takeover prompt the startup sequence is waiting on. +/// +/// The sequence holds a oneshot for exactly the duration of the prompt; a decision arriving +/// with nothing pending is a click after the fact, and it changes nothing. +#[tauri::command] +fn decide_takeover(app: tauri::AppHandle, approved: bool) { + if let Some(startup) = app.try_state::() { + startup.decide_takeover(approved); + } +} + pub fn run() { let builder = tauri::Builder::default() .plugin(tauri_plugin_single_instance::init(|app, _args, _cwd| { @@ -209,7 +221,8 @@ pub fn run() { hide_dashboard, startup_snapshot, startup_phases, - retry_startup + retry_startup, + decide_takeover ]) .setup(|app| { app.manage(AppState::new()); diff --git a/desktop/src-tauri/src/ownership.rs b/desktop/src-tauri/src/ownership.rs index a27bbf6a8e2..5b30eb366f1 100644 --- a/desktop/src-tauri/src/ownership.rs +++ b/desktop/src-tauri/src/ownership.rs @@ -16,7 +16,6 @@ //! contract that lands fills a hole rather than reshaping this file. use serde::Deserialize; -use tauri::AppHandle; /// Who a claim names. #[derive(Clone, Copy, Debug, PartialEq, Eq, Deserialize)] @@ -47,15 +46,26 @@ pub struct Claim { #[serde(tag = "kind", rename_all = "lowercase")] pub enum Recorded { /// No claim. The CLI install that registered the service owns the runtime, which is also what - /// every record written before the field existed says. - None, + /// every record written before the field existed says. The revision is the record's own + /// sequence, and a later `service claim` carries it as `expect-revision`. + None { revision: u64 }, /// A claim, whoever it names. - Owned { ownership: Claim }, + Owned { ownership: Claim, revision: u64 }, /// The claim could not be read for a decision. This is not "nobody owns it": an unreadable /// path, a corrupt anchor record and paths naming different owners all land here. Unknown { reason: String }, } +impl Default for Recorded { + /// A resolve document that carries no ownership field at all did not answer the question — + /// the older bundled CLI predates it — and an unanswered question is not a claim. + fn default() -> Self { + Self::Unknown { + reason: "the bundled CLI did not report ownership".to_owned(), + } + } +} + /// The comparison `ownershipGrantedTo` defines: same owner, same install id. /// /// True means this installation already holds consent. False against a recorded claim means a @@ -83,8 +93,8 @@ pub enum Consent { pub fn consent(recorded: &Recorded, install_id: &str) -> Consent { match recorded { Recorded::Unknown { .. } => Consent::Refuse, - Recorded::None => Consent::AskFirstTime, - Recorded::Owned { ownership } => { + Recorded::None { .. } => Consent::AskFirstTime, + Recorded::Owned { ownership, .. } => { if granted_to(Some(ownership), Owner::Desktop, install_id) { Consent::Held } else { @@ -94,35 +104,19 @@ pub fn consent(recorded: &Recorded, install_id: &str) -> Consent { } } -/// Read the recorded claim through the bundled CLI. -/// -/// Empty on purpose. Lane A publishes the machine-readable resolve the shell drives, and this is -/// the one call site that changes when it lands: it has to return the CLI's own answer, including -/// its refusals, rather than a verdict computed here. Until then the answer is *unavailable*, which -/// is not [`Recorded::None`] — the shell has not been told that nobody owns the runtime, it has not -/// asked — so no takeover is attempted and nothing is recorded. -pub fn resolve(_app: &AppHandle) -> Option { - None -} - /// One line for the startup state and for the diagnostic. -pub fn describe(recorded: Option<&Recorded>, install_id: Option<&str>) -> String { +pub fn describe(recorded: &Recorded, install_id: Option<&str>) -> String { let installation = match install_id { Some(id) => format!("installation {id}"), None => "installation id unavailable".to_owned(), }; let verdict = match (recorded, install_id) { - (None, _) => { - "recorded owner not read: the bundled CLI's resolve contract has not landed".to_owned() - } - (Some(Recorded::Unknown { reason }), _) => { + (Recorded::Unknown { reason }, _) => { format!("recorded owner could not be read ({reason}), so nothing is claimed") } - (Some(_), None) => { - "recorded owner read, but this installation has no id to compare".to_owned() - } - (Some(recorded), Some(id)) => match (consent(recorded, id), recorded) { - (Consent::Held, Recorded::Owned { ownership }) => format!( + (_, None) => "recorded owner read, but this installation has no id to compare".to_owned(), + (_, Some(id)) => match (consent(recorded, id), recorded) { + (Consent::Held, Recorded::Owned { ownership, .. }) => format!( "this installation owns the runtime (consent generation {})", ownership.consent_generation ), @@ -134,9 +128,27 @@ pub fn describe(recorded: Option<&Recorded>, install_id: Option<&str>) -> String format!("{installation}; {verdict}") } +/// Who the recorded claim names, for the consent panel. +pub fn owner_label(recorded: &Recorded) -> String { + match recorded { + Recorded::None { .. } => "no recorded owner (an npm or standalone ocx install)".to_owned(), + Recorded::Owned { ownership, .. } => match ownership.owner { + Owner::Cli => format!( + "the OpenCodex CLI install (installation {})", + ownership.install_id + ), + Owner::Desktop => format!( + "another OpenCodex desktop installation (installation {})", + ownership.install_id + ), + }, + Recorded::Unknown { reason } => format!("unknown ({reason})"), + } +} + #[cfg(test)] mod tests { - use super::{consent, describe, granted_to, Claim, Consent, Owner, Recorded}; + use super::{consent, describe, granted_to, owner_label, Claim, Consent, Owner, Recorded}; fn owned(owner: Owner, install_id: &str, generation: u64) -> Recorded { Recorded::Owned { @@ -145,6 +157,7 @@ mod tests { install_id: install_id.to_owned(), consent_generation: generation, }, + revision: 4, } } @@ -178,7 +191,10 @@ mod tests { consent(&owned(Owner::Desktop, "abc", 1), "abc"), Consent::Held ); - assert_eq!(consent(&Recorded::None, "abc"), Consent::AskFirstTime); + assert_eq!( + consent(&Recorded::None { revision: 0 }, "abc"), + Consent::AskFirstTime + ); } #[test] @@ -199,21 +215,21 @@ mod tests { reason: "a service state path could not be read".to_owned(), }; assert_eq!(consent(&unknown, "abc"), Consent::Refuse); - assert!(describe(Some(&unknown), Some("abc")).contains("could not be read")); + assert!(describe(&unknown, Some("abc")).contains("could not be read")); } #[test] fn the_wire_shape_is_the_one_the_cli_records() { let resolution: Recorded = serde_json::from_str( - r#"{"kind":"owned","ownership":{"owner":"desktop","installId":"abc","consentGeneration":3}}"#, + r#"{"kind":"owned","ownership":{"owner":"desktop","installId":"abc","consentGeneration":3},"revision":4}"#, ) .expect("the recorded resolution"); assert_eq!(resolution, owned(Owner::Desktop, "abc", 3)); assert_eq!(consent(&resolution, "abc"), Consent::Held); - assert!(describe(Some(&resolution), Some("abc")).contains("consent generation 3")); + assert!(describe(&resolution, Some("abc")).contains("consent generation 3")); assert_eq!( - serde_json::from_str::(r#"{"kind":"none"}"#).expect("no claim"), - Recorded::None + serde_json::from_str::(r#"{"kind":"none","revision":0}"#).expect("no claim"), + Recorded::None { revision: 0 } ); assert_eq!( serde_json::from_str::(r#"{"kind":"unknown","reason":"why"}"#) @@ -225,12 +241,50 @@ mod tests { } #[test] - fn the_description_separates_not_asked_from_nobody_owns_it() { - let not_asked = describe(None, Some("abc")); - let unowned = describe(Some(&Recorded::None), Some("abc")); - assert!(not_asked.contains("abc")); - assert_ne!(not_asked, unowned); - assert!(describe(None, None).contains("unavailable")); + fn a_resolve_document_without_an_ownership_answer_reads_unknown() { + assert_eq!( + Recorded::default(), + Recorded::Unknown { + reason: "the bundled CLI did not report ownership".to_owned() + } + ); + assert!(serde_json::from_str::(r#"{"kind":"none"}"#).is_err()); + } + + #[test] + fn the_description_separates_not_read_from_nobody_owns_it() { + let unread = describe( + &Recorded::Unknown { + reason: "why".to_owned(), + }, + Some("abc"), + ); + let unowned = describe(&Recorded::None { revision: 0 }, Some("abc")); + assert!(unread.contains("abc")); + assert_ne!(unread, unowned); + assert!(describe(&Recorded::None { revision: 0 }, None).contains("unavailable")); + } + + #[test] + fn owner_label_names_who_the_claim_is_for() { + assert_eq!( + owner_label(&Recorded::None { revision: 0 }), + "no recorded owner (an npm or standalone ocx install)" + ); + assert_eq!( + owner_label(&owned(Owner::Cli, "npm-1", 1)), + "the OpenCodex CLI install (installation npm-1)" + ); + assert_eq!( + owner_label(&owned(Owner::Desktop, "other", 2)), + "another OpenCodex desktop installation (installation other)" + ); + assert_eq!( + owner_label(&Recorded::Unknown { + reason: "why".to_owned() + }), + "unknown (why)" + ); } #[test] diff --git a/desktop/src-tauri/src/resolve.rs b/desktop/src-tauri/src/resolve.rs index 8172006d0ff..e644b6edcc4 100644 --- a/desktop/src-tauri/src/resolve.rs +++ b/desktop/src-tauri/src/resolve.rs @@ -14,6 +14,7 @@ //! reading that must never happen is "the resolve failed, so nobody must be listening". use crate::endpoint::ProxyEndpoint; +use crate::ownership::Recorded; use serde::Deserialize; use std::path::PathBuf; use tauri::AppHandle; @@ -52,6 +53,36 @@ pub struct Port { pub configured: u16, } +/// Whether the CLI says a desktop takeover can be offered. +/// +/// The token is the binding a later `ocx service claim` repeats back: it covers the exact +/// subject and managing-CLI observations the consent was approved against, so a claim made +/// after either moved is refused rather than recorded. +#[derive(Clone, Debug, PartialEq, Eq, Deserialize)] +#[serde(tag = "kind", rename_all = "camelCase")] +pub enum Takeover { + #[serde(rename_all = "camelCase")] + Supported { + protocol_version: u64, + minimum_cli_version: String, + token: String, + }, + Blocked { + reason: String, + detail: String, + }, +} + +impl Default for Takeover { + /// An older bundled CLI carries no takeover answer at all; silence is not approval. + fn default() -> Self { + Self::Blocked { + reason: "unreported".to_owned(), + detail: "the bundled CLI did not report takeover compatibility".to_owned(), + } + } +} + #[derive(Clone, Debug, PartialEq, Eq, Deserialize)] #[serde(rename_all = "camelCase")] pub struct Resolved { @@ -60,6 +91,12 @@ pub struct Resolved { pub config_home: String, pub port: Port, pub liveness: Liveness, + /// The recorded runtime owner, already in the CLI's three answers. Absent on older + /// documents, which read as unknown rather than as nobody owning the runtime. + #[serde(default)] + pub ownership: Recorded, + #[serde(default)] + pub takeover: Takeover, } impl Resolved { @@ -228,8 +265,10 @@ pub async fn run(app: &AppHandle, deadline: Instant) -> Resolution { #[cfg(test)] mod tests { use super::{ - live_verdict, loopback_reachable, may_start, read, LiveVerdict, Resolution, Status, SCHEMA, + live_verdict, loopback_reachable, may_start, read, LiveVerdict, Resolution, Status, + Takeover, SCHEMA, }; + use crate::ownership::{Owner, Recorded}; const LIVE: &str = r#"{"schema":"ocx-resolve/1","cliVersion":"2.61.0","configHome":"/h", "port":{"effective":10100,"configured":10100,"source":"runtime-record"}, @@ -318,6 +357,69 @@ mod tests { } } + #[test] + fn ownership_and_takeover_answers_are_read_whole() { + let document = format!( + "{}{}}}", + LIVE.strip_suffix('}').unwrap(), + r#","ownership":{"kind":"owned","ownership":{"owner":"cli","installId":"npm-1","consentGeneration":2},"revision":9},"takeover":{"kind":"supported","protocolVersion":1,"minimumCliVersion":"2.61.0","token":"abc"}"# + ); + let resolution = read(Some(0), document.as_bytes(), b""); + let resolved = match resolution.resolved() { + Some(resolved) => resolved.clone(), + None => panic!("{}", resolution.reason().unwrap()), + }; + assert_eq!( + resolved.ownership, + Recorded::Owned { + ownership: crate::ownership::Claim { + owner: Owner::Cli, + install_id: "npm-1".to_owned(), + consent_generation: 2, + }, + revision: 9, + } + ); + assert!(matches!( + resolved.takeover, + Takeover::Supported { ref token, .. } if token == "abc" + )); + } + + #[test] + fn a_missing_ownership_or_takeover_answer_is_not_consent() { + // Older bundled CLIs carry neither field; silence must read unknown/blocked, never + // "nobody owns it" or "takeover supported". + let resolved = read(Some(0), LIVE.as_bytes(), b"") + .resolved() + .expect("a document") + .clone(); + assert!(matches!(resolved.ownership, Recorded::Unknown { .. })); + assert!(matches!(resolved.takeover, Takeover::Blocked { .. })); + assert_eq!(resolved.takeover, Takeover::default()); + } + + #[test] + fn a_blocked_takeover_carries_its_reason() { + let document = format!( + "{}{}}}", + LIVE.strip_suffix('}').unwrap(), + r#","ownership":{"kind":"none","revision":0},"takeover":{"kind":"blocked","reason":"managing-cli-unsupported","detail":"path uses 2.59.0","minimumCliVersion":"2.61.0"}"# + ); + let resolved = read(Some(0), document.as_bytes(), b"") + .resolved() + .expect("a document") + .clone(); + assert_eq!(resolved.ownership, Recorded::None { revision: 0 }); + assert_eq!( + resolved.takeover, + Takeover::Blocked { + reason: "managing-cli-unsupported".to_owned(), + detail: "path uses 2.59.0".to_owned(), + } + ); + } + #[test] fn a_schema_this_app_does_not_know_is_unknown() { let future = LIVE.replace("ocx-resolve/1", "ocx-resolve/2"); diff --git a/desktop/src-tauri/src/startup.rs b/desktop/src-tauri/src/startup.rs index 0fccc6fddfc..bfa7c7a9f2e 100644 --- a/desktop/src-tauri/src/startup.rs +++ b/desktop/src-tauri/src/startup.rs @@ -17,11 +17,12 @@ use crate::{ auth::Auth, + claim, endpoint::ProxyEndpoint, first_run::{self, StartAtLogin}, identity, ownership, proxy::{ProxyClient, RuntimeIdentity}, - resolve, + resolve, runtime_stop, sidecar::{self, SidecarWatch}, tray_availability::{self, TrayAvailability}, AppState, @@ -35,6 +36,7 @@ use std::{ }, }; use tauri::{AppHandle, Emitter, Manager}; +use tokio::sync::oneshot; use tokio::time::{sleep, sleep_until, Duration, Instant}; /// The event the bootstrap page listens on. @@ -110,6 +112,7 @@ pub enum Phase { Resolving, Probing, Attaching, + TakingOver, Starting, Waiting, Ready, @@ -121,11 +124,12 @@ pub enum Phase { /// /// [`Phase::NotStarted`] is absent on purpose. It is the state of not having run, so a checklist /// row for it would be a step that never completes. -pub const PHASES: [Phase; 8] = [ +pub const PHASES: [Phase; 9] = [ Phase::Registering, Phase::Resolving, Phase::Probing, Phase::Attaching, + Phase::TakingOver, Phase::Starting, Phase::Waiting, Phase::Ready, @@ -141,6 +145,7 @@ impl Phase { Self::Resolving => "resolving", Self::Probing => "probing", Self::Attaching => "attaching", + Self::TakingOver => "taking-over", Self::Starting => "starting", Self::Waiting => "waiting", Self::Ready => "ready", @@ -155,6 +160,7 @@ impl Phase { Self::Resolving => "Resolving the configuration home and port", Self::Probing => "Looking for a runtime that is already listening", Self::Attaching => "Attaching to the runtime that answered", + Self::TakingOver => "Taking over the runtime that was already listening", Self::Starting => "Starting the bundled runtime", Self::Waiting => "Waiting for the runtime to report healthy", Self::Ready => "Ready", @@ -213,6 +219,21 @@ pub struct Progress { pub dashboard: Option, pub diagnostic: Option, pub can_retry: bool, + /// Present only while the shell is waiting on the user's takeover decision. + pub consent: Option, +} + +/// What the consent panel renders. `blocked` carries the CLI's refusal reason when a +/// takeover cannot be offered; the panel is shown only for the offerable case today, but +/// the field is part of the wire so a later UI does not need a schema change. +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct ConsentPrompt { + pub endpoint: String, + pub port: u16, + pub home: String, + pub owner: String, + pub blocked: Option, } impl Progress { @@ -227,6 +248,7 @@ impl Progress { dashboard: None, diagnostic: None, can_retry: phase == Phase::Failed, + consent: None, } } } @@ -263,19 +285,85 @@ struct Target { #[derive(Clone, Debug)] pub struct Registration { pub login: StartAtLogin, - /// This installation's own id, and what the recorded runtime owner says about it. - pub identity: String, + /// This installation's own id, minted once in the app's config directory. + pub install_id: Option, } struct Live { latest: Progress, reported: Vec<&'static str>, + /// The takeover decision state. Answered stays set until the run clears it: the deadline + /// extension lands between the decision and the clear, and the guard must keep waiting + /// through both. + consent: ConsentState, + /// The current run's ceiling. + /// + /// The consent wait moves it by however long the person took, so the deadline guard + /// re-reads it instead of racing a stale copy. It lives under this lock so the expiry + /// decision and the terminal publish are one critical section against consent transitions. + deadline: Instant, +} + +impl Live { + fn is_settled(&self) -> bool { + self.latest.phase == Phase::Ready.id() || self.latest.phase == Phase::Failed.id() + } + + /// Record the latest state. A run that already said how it ended refuses further + /// reports: the terminal state is the page's promise that the screen stopped changing, + /// and a probe resuming after the expiry landed must not move it back — nor reopen the + /// consent gate that reads this state. Returns whether the report was taken. + fn publish(&mut self, progress: &mut Progress, failed_in: Option) -> bool { + if self.is_settled() { + return false; + } + if !self.reported.contains(&progress.phase) + && progress.phase != Phase::Ready.id() + && progress.phase != Phase::Failed.id() + { + self.reported.push(progress.phase); + } + progress.completed = self + .reported + .iter() + .copied() + .filter(|id| *id != progress.phase) + .collect(); + progress.failed_phase = failed_in.map(Phase::id); + self.latest = progress.clone(); + true + } +} + +/// The takeover prompt's decision state. +enum ConsentState { + /// No prompt is up and none was just answered. + Idle, + /// A prompt is up; the sender resolves with the user's decision. + Pending(oneshot::Sender), + /// The user answered and the run has not yet consumed the extension. + Answered, +} + +/// What the deadline guard's expiry step found. +enum Expiry { + /// The run is terminal or superseded; the guard is done. + Dead, + /// A consent prompt is pending or its answer is being consumed; re-check shortly. + Blocked, + /// Not expired yet; the current ceiling plus its grace. + Waiting(Instant), + /// Expired and the failure was published in the same critical section; emit it. + Fired(Box), } /// The sequence's managed state: the latest thing it said, what it has already finished, and /// whether it is running, so a retry cannot start a second run alongside the first. pub struct Startup { live: Mutex, + /// Serialize state publication with its synchronous event dispatch. Always acquired + /// before `live`, and never held across an await. + reporting: Mutex<()>, running: AtomicBool, /// Which run the state belongs to. /// @@ -292,7 +380,10 @@ impl Startup { live: Mutex::new(Live { latest: Progress::new(Phase::NotStarted, 0), reported: Vec::new(), + consent: ConsentState::Idle, + deadline: Instant::now(), }), + reporting: Mutex::new(()), running: AtomicBool::new(false), generation: AtomicU64::new(0), registered: Mutex::new(None), @@ -322,8 +413,51 @@ impl Startup { self.live().latest.clone() } + /// The user's answer to a pending takeover prompt. Nothing pending is a no-op: a retry + /// or a late click must never be read as a decision for a prompt that is not up. + pub fn decide_takeover(&self, approved: bool) { + let mut live = self.live(); + match std::mem::replace(&mut live.consent, ConsentState::Idle) { + ConsentState::Pending(sender) => { + live.consent = ConsentState::Answered; + let _ = sender.send(approved); + } + // A late click or a duplicate decision answers nothing: restore what was there. + prior => live.consent = prior, + } + } + + /// Register the pending prompt, unless the run already ended. A guard expiry can win the + /// race against the prompt being posted; posting one anyway would leave a receiver that + /// waits forever on a decision nobody can see. + fn await_consent(&self) -> Option> { + let mut live = self.live(); + if live.is_settled() { + return None; + } + let (sender, receiver) = oneshot::channel(); + live.consent = ConsentState::Pending(sender); + Some(receiver) + } + + /// Consume the decision and publish the extended ceiling in the same critical section, so + /// the guard's next expiry check sees either a pending/answered prompt or the new deadline, + /// never the gap between them. + fn resolve_consent(&self, deadline: Instant) { + let mut live = self.live(); + live.deadline = deadline; + live.consent = ConsentState::Idle; + } + + fn set_deadline(&self, deadline: Instant) { + self.live().deadline = deadline; + } + fn restart(&self) { + // A retry during a pending consent prompt drops the sender, so the waiting run reads + // the decision as declined rather than pairing an old prompt with a new sequence. let mut live = self.live(); + live.consent = ConsentState::Idle; live.reported.clear(); live.latest = Progress::new(Phase::NotStarted, 0); } @@ -332,27 +466,88 @@ impl Startup { /// /// A terminal state is the page's only promise that the screen has stopped changing, so it is /// also what tells a late guard there is nothing left to report. + #[cfg(test)] fn settled(&self) -> bool { - let phase = self.live().latest.phase; - phase == Phase::Ready.id() || phase == Phase::Failed.id() + self.live().is_settled() } - fn publish(&self, progress: &mut Progress, failed_in: Option) { + fn publish(&self, progress: &mut Progress, failed_in: Option) -> bool { let mut live = self.live(); - if !live.reported.contains(&progress.phase) - && progress.phase != Phase::Ready.id() - && progress.phase != Phase::Failed.id() - { - live.reported.push(progress.phase); + live.publish(progress, failed_in) + } + + fn with_reporting(&self, report: impl FnOnce() -> T) -> T { + let _reporting = self + .reporting + .lock() + .unwrap_or_else(PoisonError::into_inner); + report() + } + + /// Publish a terminal state for a run that did not report one itself. + /// + /// Idempotent and bound to the run it was started for: a run that already said Ready or + /// Failed is left alone, and a caller whose run has been superseded by a retry says + /// nothing. The check and the publish are one critical section, so no other reporter can + /// slip a state between them. + fn settle(&self, started: Instant, generation: u64, reason: String) -> Option { + let mut live = self.live(); + self.settle_locked(&mut live, started, generation, reason) + } + + fn settle_locked( + &self, + live: &mut Live, + started: Instant, + generation: u64, + reason: String, + ) -> Option { + if self.generation.load(Ordering::Acquire) != generation || live.is_settled() { + return None; + } + let stalled_in = live.latest.phase; + let elapsed_ms = elapsed(started); + let mut progress = Progress::new(Phase::Failed, elapsed_ms); + progress.diagnostic = Some( + [ + format!( + "OpenCodex desktop {} on {}", + env!("CARGO_PKG_VERSION"), + std::env::consts::OS + ), + format!("state: {stalled_in}"), + format!("reason: {reason}"), + format!("elapsed: {elapsed_ms}ms"), + ] + .join("\n"), + ); + progress.detail = Some(reason); + live.publish(&mut progress, Phase::from_id(stalled_in)); + Some(progress) + } + + /// The deadline guard's atomic expiry step. The deadline read, the consent state, the + /// terminal check and the failure publish all share one critical section, so a prompt + /// posted or an answer consumed on the other side of the lock can never meet a failure + /// already in flight. + fn expire_run(&self, started: Instant, generation: u64, reason: String) -> Expiry { + let mut live = self.live(); + if self.generation.load(Ordering::Acquire) != generation || live.is_settled() { + return Expiry::Dead; + } + if !matches!(live.consent, ConsentState::Idle) { + // A prompt is up or an answer is being consumed. The budget does not run + // against the person, so there is nothing to expire. + return Expiry::Blocked; + } + let wake = live.deadline + SETTLE_GRACE; + if wake > Instant::now() { + return Expiry::Waiting(wake); + } + match self.settle_locked(&mut live, started, generation, reason) { + Some(progress) => Expiry::Fired(Box::new(progress)), + None => Expiry::Dead, } - progress.completed = live - .reported - .iter() - .copied() - .filter(|id| *id != progress.phase) - .collect(); - progress.failed_phase = failed_in.map(Phase::id); - live.latest = progress.clone(); } } @@ -377,6 +572,7 @@ pub fn begin(app: &AppHandle) { startup.restart(); let generation = startup.generation.fetch_add(1, Ordering::AcqRel) + 1; let started = Instant::now(); + startup.set_deadline(started + DEADLINE); let app = app.clone(); // The ceiling is a promise to the page, and something has to keep it when the run does not. @@ -386,16 +582,41 @@ pub fn begin(app: &AppHandle) { // cannot tell from a hung application, which is the whole thing this surface exists to avoid. let guard = app.clone(); tauri::async_runtime::spawn(async move { - sleep_until(started + DEADLINE + SETTLE_GRACE).await; - settle( - &guard, - started, - generation, - format!( - "the startup sequence did not finish within {} seconds", - DEADLINE.as_secs() - ), - ); + // The consent wait extends the shared deadline, and while a prompt is up the budget + // does not run at all. The expiry check, the consent state and the terminal publish + // share one critical section, so a prompt posted or an answer consumed can never meet + // a failure already in flight. + loop { + let Some(startup) = guard.try_state::() else { + return; + }; + let expiry = startup.with_reporting(|| { + let expiry = startup.expire_run( + started, + generation, + format!( + "the startup sequence did not finish within {} seconds", + DEADLINE.as_secs() + ), + ); + if let Expiry::Fired(progress) = &expiry { + let _ = guard.emit(PHASE_EVENT, progress); + } + expiry + }); + match expiry { + Expiry::Dead => return, + Expiry::Blocked => { + sleep(POLL).await; + continue; + } + Expiry::Waiting(wake) => { + sleep_until(wake).await; + continue; + } + Expiry::Fired(_) => return, + } + } }); tauri::async_runtime::spawn(async move { @@ -420,31 +641,15 @@ fn settle(app: &AppHandle, started: Instant, generation: u64, reason: String) { let Some(startup) = app.try_state::() else { return; }; - if startup.generation.load(Ordering::Acquire) != generation || startup.settled() { - return; - } - let stalled_in = startup.latest().phase; - let elapsed_ms = elapsed(started); - let mut progress = Progress::new(Phase::Failed, elapsed_ms); - progress.diagnostic = Some( - [ - format!( - "OpenCodex desktop {} on {}", - env!("CARGO_PKG_VERSION"), - std::env::consts::OS - ), - format!("state: {stalled_in}"), - format!("reason: {reason}"), - format!("elapsed: {elapsed_ms}ms"), - ] - .join("\n"), - ); - progress.detail = Some(reason); - emit(app, progress, Phase::from_id(stalled_in)); + startup.with_reporting(|| { + if let Some(progress) = startup.settle(started, generation, reason) { + let _ = app.emit(PHASE_EVENT, progress); + } + }); } async fn run(app: &AppHandle, started: Instant) { - let deadline = started + DEADLINE; + let mut deadline = started + DEADLINE; // Publishing comes before any lookup that can fail. A sequence that returns before it has // said anything leaves the page unable to tell "not started" from "still going". report(app, started, Phase::Registering, None); @@ -456,11 +661,7 @@ async fn run(app: &AppHandle, started: Instant) { app, started, Phase::Registering, - Some(format!( - "{}; {}", - registration.login.describe(), - registration.identity - )), + Some(registration.login.describe().to_owned()), ); report(app, started, Phase::Resolving, None); @@ -514,10 +715,11 @@ async fn run(app: &AppHandle, started: Instant) { started, Phase::Resolving, Some(format!( - "{} with a configuration home of {}, resolved by the bundled CLI {}", + "{} with a configuration home of {}, resolved by the bundled CLI {}; {}", target.endpoint.url(""), target.home.display(), - answer.cli_version + answer.cli_version, + ownership::describe(&answer.ownership, registration.install_id.as_deref()) )), ); @@ -532,29 +734,100 @@ async fn run(app: &AppHandle, started: Instant) { } }), ); + let mut took_over = false; match resolve::live_verdict(&resolution) { resolve::LiveVerdict::Attach => { - report( - app, - started, - Phase::Attaching, - Some("a runtime was already listening, so this app is a guest on it".to_owned()), - ); - if bind(app, &proxy, deadline).await.is_none() { - fail( - app, - started, - Some(&target), - ®istration, - &watch, - Phase::Attaching, - "the runtime answered but did not identify itself, so this app did not attach" - .to_owned(), - ); - return; + // Without our own id nothing can ever match us, which is the answer Refuse gives. + let consent = match registration.install_id.as_deref() { + Some(install_id) => ownership::consent(&answer.ownership, install_id), + None => ownership::Consent::Refuse, + }; + match attach_plan(consent, &answer.takeover) { + AttachPlan::Guest(detail) => { + attach_as_guest( + app, + started, + &target, + ®istration, + &watch, + &proxy, + endpoint, + deadline, + detail, + ) + .await; + return; + } + AttachPlan::Ask(token) => { + // The prompt has to be visible even when this launch started hidden. + if let Some(window) = app.get_webview_window("main") { + crate::window::show(&window); + } + let Some(startup) = app.try_state::() else { + return; + }; + let Some(receiver) = startup.await_consent() else { + // The run already ended (an expiry won the race to the terminal + // state). Posting the prompt now would wait on a decision nobody + // can see, so the run stops here instead. + return; + }; + let mut progress = Progress::new(Phase::Attaching, elapsed(started)); + progress.detail = Some( + "a runtime was already listening; waiting for a decision on taking it over" + .to_owned(), + ); + progress.consent = Some(ConsentPrompt { + endpoint: target.endpoint.url(""), + port: target.endpoint.port, + home: target.home.display().to_string(), + owner: ownership::owner_label(&answer.ownership), + blocked: None, + }); + emit(app, progress, None); + // The user may take any time; the budget exists to bound the machinery, not + // the person, so the deadline moves by whatever the decision took. + let asked = Instant::now(); + let approved = receiver.await.unwrap_or(false); + deadline += asked.elapsed(); + // The extension and the clear are one critical section: the guard sees + // either a prompt still pending or the moved ceiling, never the gap. + startup.resolve_consent(deadline); + if !approved { + attach_as_guest( + app, + started, + &target, + ®istration, + &watch, + &proxy, + endpoint, + deadline, + "a runtime was already listening and taking it over was declined, so this app is a guest on it" + .to_owned(), + ) + .await; + return; + } + if take_over( + app, + started, + &mut deadline, + &target, + ®istration, + &watch, + &proxy, + &answer.ownership, + &token, + ) + .await + .is_err() + { + return; + } + took_over = true; + } } - finish(app, started, endpoint); - return; } // Something holds the port and this app cannot manage it. That is not an absence, so it // does not authorise starting a second runtime beside it either. @@ -572,8 +845,9 @@ async fn run(app: &AppHandle, started: Instant) { } resolve::LiveVerdict::NotLive => {} } - if !resolve::may_start(&resolution) { - // Only a proven absence authorises a start. Nothing else may fall through to one. + if !took_over && !resolve::may_start(&resolution) { + // Only a proven absence authorises a start. Nothing else may fall through to one. A + // takeover just proved its own absence by stopping what was there. fail( app, started, @@ -672,6 +946,190 @@ async fn run(app: &AppHandle, started: Instant) { ); } +/// What an attach turns into once the recorded owner and the CLI's compatibility answer are +/// laid next to each other. `Ask` carries the token the claim has to be made against. +enum AttachPlan { + /// Stay a guest on what answered; the string is the detail the phase reports. + Guest(String), + /// Offer the takeover and wait on the user. + Ask(String), +} + +fn attach_plan(consent: ownership::Consent, takeover: &resolve::Takeover) -> AttachPlan { + match consent { + ownership::Consent::Held => AttachPlan::Guest( + "a runtime was already listening and this installation already owns it".to_owned(), + ), + ownership::Consent::Refuse => AttachPlan::Guest( + "a runtime was already listening; its recorded owner could not be read, so this app is a guest on it and asked nothing".to_owned(), + ), + ownership::Consent::AskFirstTime | ownership::Consent::AskAgain => match takeover { + resolve::Takeover::Blocked { reason, detail } => AttachPlan::Guest(format!( + "a runtime was already listening, but taking it over is not available ({reason}: {detail}), so this app is a guest on it" + )), + resolve::Takeover::Supported { token, .. } => AttachPlan::Ask(token.clone()), + }, + } +} + +/// Report, bind and finish as a guest on the runtime that answered. +#[allow(clippy::too_many_arguments)] +async fn attach_as_guest( + app: &AppHandle, + started: Instant, + target: &Target, + registration: &Registration, + watch: &SidecarWatch, + proxy: &ProxyClient, + endpoint: ProxyEndpoint, + deadline: Instant, + detail: String, +) { + report(app, started, Phase::Attaching, Some(detail)); + if bind(app, proxy, deadline).await.is_none() { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::Attaching, + "the runtime answered but did not identify itself, so this app did not attach" + .to_owned(), + ); + return; + } + finish(app, started, endpoint); +} + +/// Stop the runtime that answered, wait for its silence, and record this installation as +/// the owner. An `Err` has already been reported; `Ok` means the Starting branch may run. +#[allow(clippy::too_many_arguments)] +async fn take_over( + app: &AppHandle, + started: Instant, + deadline: &mut Instant, + target: &Target, + registration: &Registration, + watch: &SidecarWatch, + proxy: &ProxyClient, + recorded: &ownership::Recorded, + token: &str, +) -> Result<(), ()> { + report( + app, + started, + Phase::TakingOver, + Some("stopping the runtime that was already listening".to_owned()), + ); + let stopped = runtime_stop::run(app, *deadline).await; + // A refused connection, not exit 0 and not the probe's deadline, is the receipt: `ocx stop` + // reports exit 79 when the proxy stopped but history cleanup failed after it exited, and a + // `None` from alive_within is only the clock running out — neither is silence. So whatever + // the stop reported, the claim is made only once the port actively refuses. + let mut silent = false; + let mut still_answering = false; + while Instant::now() < *deadline { + match proxy.alive_within(*deadline).await { + Some(Err(_)) => { + silent = true; + break; + } + Some(Ok(_)) => { + still_answering = true; + sleep(POLL).await; + } + None => { + still_answering = false; + break; + } + } + } + if !silent { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + format!( + "the runtime that was already listening {} ({})", + if still_answering { + "is still answering after the stop" + } else { + "did not go silent before the deadline" + }, + stopped.describe() + ), + ); + return Err(()); + } + if !stopped.is_stopped() { + report( + app, + started, + Phase::TakingOver, + Some(format!( + "the runtime that was already listening stopped answering (stop reported: {})", + stopped.describe() + )), + ); + } + + report( + app, + started, + Phase::TakingOver, + Some("recording this installation as the runtime owner".to_owned()), + ); + let install_id = registration.install_id.clone().unwrap_or_default(); + // An unknown record reaches here only off the UI path, and the claim has to refuse rather + // than fabricate the subject it is claiming against. + let Some(argv) = claim::args(&install_id, recorded, token) else { + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + "the recorded owner could not be read, so no claim was made".to_owned(), + ); + return Err(()); + }; + match claim::run(app, argv, *deadline).await { + claim::ClaimResult::Recorded(ownership) => { + report( + app, + started, + Phase::TakingOver, + Some(format!( + "this installation now owns the runtime (consent generation {})", + ownership.consent_generation + )), + ); + Ok(()) + } + claim::ClaimResult::Failed(message) => { + // The runtime is stopped either way. Refusing here leaves the next launch an + // ordinary absence to start into, which is the acceptable end state. + fail( + app, + started, + Some(target), + registration, + watch, + Phase::TakingOver, + format!( + "the runtime was stopped, but this installation could not be recorded as its owner: {message}" + ), + ); + Err(()) + } + } +} + /// Establish the app's own surface: the tray verdict, the tray, and the login item. /// /// It happens once per process. A retry re-runs the runtime half of the sequence, and running this @@ -723,10 +1181,11 @@ async fn register(app: &AppHandle, deadline: Instant) -> Registration { // This installation's own id, and what the recorded runtime owner says about it. The claim // lives in the shared service install state and the CLI is what reads it; the comparison // against our own id is the rule that record publishes. - let install_id = identity::install_id(app); + // This installation's own id; what the recorded runtime owner says about it is part of the + // resolve answer, so the identity line is written where the answer exists. let registration = Registration { login, - identity: ownership::describe(ownership::resolve(app).as_ref(), install_id.as_deref()), + install_id: identity::install_id(app), }; if let Some(startup) = app.try_state::() { startup.remember_registration(registration.clone()); @@ -826,7 +1285,11 @@ fn finish(app: &AppHandle, started: Instant, endpoint: ProxyEndpoint) { let dashboard = endpoint.url("/#/usage"); let mut progress = Progress::new(Phase::Ready, elapsed(started)); progress.dashboard = Some(dashboard.clone()); - emit(app, progress, None); + if !emit(app, progress, None) { + // The run already ended — the expiry won while this one was still binding. The + // terminal state stays and the window must not navigate away from it. + return; + } if let Some(window) = app.get_webview_window("main") { // justified: replacing the bootstrap page with the dashboard is how this window has always // navigated, and the string is a URL this process resolved, not anything a page supplied. @@ -890,7 +1353,10 @@ pub fn diagnostic( None => lines.push("endpoint: not resolved".to_owned()), } lines.push(format!("start at login: {}", registration.login.describe())); - lines.push(format!("runtime ownership: {}", registration.identity)); + lines.push(format!( + "installation id: {}", + registration.install_id.as_deref().unwrap_or("not minted") + )); lines.push(match watch.exit() { Some(exit) => format!("runtime process: {}", exit.describe()), None => "runtime process: still running or never started".to_owned(), @@ -911,11 +1377,21 @@ fn report(app: &AppHandle, started: Instant, phase: Phase, detail: Option) { +/// Publish and emit one state. A report refused because the run already ended is not +/// emitted either, so a stale event cannot move the page past the terminal state the +/// snapshot keeps. Returns whether the report was published. +fn emit(app: &AppHandle, mut progress: Progress, failed_in: Option) -> bool { if let Some(startup) = app.try_state::() { - startup.publish(&mut progress, failed_in); + return startup.with_reporting(|| { + if !startup.publish(&mut progress, failed_in) { + return false; + } + let _ = app.emit(PHASE_EVENT, progress); + true + }); } let _ = app.emit(PHASE_EVENT, progress); + true } fn elapsed(started: Instant) -> u64 { @@ -925,11 +1401,65 @@ fn elapsed(started: Instant) -> u64 { #[cfg(test)] mod tests { use super::{ - shows_window, unavailable, LaunchOrigin, Phase, Progress, Startup, AUTOSTART_FLAG, - DEADLINE, PHASES, POLL, + attach_plan, shows_window, unavailable, AttachPlan, ConsentState, Expiry, LaunchOrigin, + Phase, Progress, Startup, AUTOSTART_FLAG, DEADLINE, PHASES, POLL, }; + use crate::ownership::Consent; + use crate::resolve::Takeover; use crate::tray_availability::TrayAvailability; - use tokio::time::Duration; + use std::sync::atomic::Ordering; + use tokio::time::{Duration, Instant}; + + fn supported() -> Takeover { + Takeover::Supported { + protocol_version: 1, + minimum_cli_version: "2.61.0".to_owned(), + token: "tok".to_owned(), + } + } + + fn blocked() -> Takeover { + Takeover::Blocked { + reason: "managing-cli-unsupported".to_owned(), + detail: "path uses 2.59.0".to_owned(), + } + } + + #[test] + fn a_retry_drops_a_prompt_the_run_is_still_waiting_on() { + // The waiting run reads the dropped sender as declined, so a stale prompt can never + // pair a decision meant for it with the retried sequence. + let startup = Startup::new(); + let mut receiver = startup.await_consent().expect("no terminal state yet"); + startup.restart(); + assert!(matches!( + receiver.try_recv(), + Err(tokio::sync::oneshot::error::TryRecvError::Closed) + )); + } + + #[test] + fn an_ask_only_arises_when_the_takeover_can_be_taken() { + // Held and Refuse never ask, whatever the CLI reported about compatibility. + assert!(matches!( + attach_plan(Consent::Held, &supported()), + AttachPlan::Guest(_) + )); + assert!(matches!( + attach_plan(Consent::Refuse, &supported()), + AttachPlan::Guest(_) + )); + match attach_plan(Consent::AskFirstTime, &supported()) { + AttachPlan::Ask(token) => assert_eq!(token, "tok"), + AttachPlan::Guest(detail) => panic!("{detail}"), + } + match attach_plan(Consent::AskAgain, &blocked()) { + AttachPlan::Guest(detail) => { + assert!(detail.contains("managing-cli-unsupported: path uses 2.59.0")) + } + AttachPlan::Ask(_) => panic!("a blocked takeover is not an offer"), + } + } #[test] fn not_having_started_is_not_a_step_of_the_run() { @@ -1047,4 +1577,150 @@ mod tests { .all(|budget| *budget <= Duration::from_secs(45))); assert!(POLL < DEADLINE); } + + #[test] + fn an_expired_run_publishes_failed_in_the_same_step() { + // The expiry decision and the terminal publish share one critical section: an expired + // deadline with no prompt up fails the run, and the failure is already there when the + // call returns. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + match startup.expire_run( + tokio::time::Instant::now() - Duration::from_secs(60), + 1, + "expired".to_owned(), + ) { + Expiry::Fired(progress) => { + assert_eq!(progress.phase, Phase::Failed.id()); + assert!(progress.can_retry); + } + _ => panic!("an expired deadline with no consent must fire"), + } + assert!(startup.settled()); + // A second expiry for the same run says nothing. + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "again".to_owned()), + Expiry::Dead + )); + } + + #[test] + fn a_pending_prompt_blocks_expiry_and_the_prompt_still_resolves() { + // The losing side of the race the guard used to win: the prompt is up while the + // deadline sits in the past. Expiry must yield, and the user's answer must still + // reach the waiting run. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + let mut receiver = startup.await_consent().expect("no terminal state yet"); + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Blocked + )); + startup.decide_takeover(true); + assert_eq!(receiver.try_recv(), Ok(true)); + // The answer was consumed but the extension has not landed yet: still not expirable. + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Blocked + )); + // Once the run publishes the moved ceiling the guard waits on it instead of firing. + startup.resolve_consent(tokio::time::Instant::now() + Duration::from_secs(60)); + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Waiting(_) + )); + } + + #[test] + fn a_terminal_run_posts_no_prompt() { + // The other half of the race: the failure already landed, so the ask path must not + // register a prompt that would wait on a decision nobody can see. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + let mut terminal = Progress::new(Phase::Failed, 1); + startup.publish(&mut terminal, None); + assert!(startup.await_consent().is_none()); + // And the consent state stays idle, so a later run is not shadowed by a stale prompt. + assert!(matches!(startup.live().consent, ConsentState::Idle)); + } + + #[test] + fn a_terminal_state_is_not_moved_by_a_late_report() { + // The expiry lands while the run is still inside a probe; the probe then resumes and + // reports. Neither the snapshot nor the consent gate may move: the terminal state is + // the page's promise that it stopped changing, and a report that could undo it would + // also reopen the prompt the terminal state just ruled out. + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + match startup.expire_run( + tokio::time::Instant::now() - Duration::from_secs(60), + 1, + "expired".to_owned(), + ) { + Expiry::Fired(progress) => assert_eq!(progress.phase, Phase::Failed.id()), + _ => panic!("an expired deadline with no consent must fire"), + } + let mut late = Progress::new(Phase::Probing, 2); + assert!(!startup.publish(&mut late, None)); + assert_eq!(startup.latest().phase, Phase::Failed.id()); + assert!(startup.await_consent().is_none()); + // A second terminal report is refused as well: the first ending stands. + let mut ready = Progress::new(Phase::Ready, 3); + assert!(!startup.publish(&mut ready, None)); + assert_eq!(startup.latest().phase, Phase::Failed.id()); + } + + #[test] + fn expiry_waits_for_an_accepted_report_to_be_dispatched() { + let startup = Startup::new(); + startup.generation.store(1, Ordering::SeqCst); + startup.set_deadline(Instant::now() - Duration::from_secs(60)); + let events = std::sync::Mutex::new(Vec::new()); + let (checked_tx, checked_rx) = std::sync::mpsc::channel(); + std::thread::scope(|scope| { + startup.with_reporting(|| { + let mut progress = Progress::new(Phase::Probing, 1); + assert!(startup.publish(&mut progress, None)); + let startup = &startup; + let events = &events; + scope.spawn(move || { + // The report was accepted but has not dispatched yet. Expiry cannot + // overtake it, even though the state mutex itself is no longer held. + assert!(matches!( + startup.reporting.try_lock(), + Err(std::sync::TryLockError::WouldBlock) + )); + checked_tx.send(()).unwrap(); + startup.with_reporting(|| { + match startup.expire_run(Instant::now(), 1, "expired".to_owned()) { + Expiry::Fired(progress) => events.lock().unwrap().push(progress.phase), + _ => panic!("the unblocked expiry must publish failure"), + } + }); + }); + checked_rx.recv().unwrap(); + events.lock().unwrap().push(progress.phase); + }); + }); + assert_eq!( + *events.lock().unwrap(), + vec![Phase::Probing.id(), Phase::Failed.id()] + ); + } + + #[test] + fn a_superseded_guard_reports_nothing() { + // A retry bumped the generation: the old guard's expiry is dead even with the + // deadline in the past. + let startup = Startup::new(); + startup.generation.store(2, Ordering::SeqCst); + startup.set_deadline(tokio::time::Instant::now() - Duration::from_secs(60)); + assert!(matches!( + startup.expire_run(tokio::time::Instant::now(), 1, "expired".to_owned()), + Expiry::Dead + )); + } } diff --git a/desktop/ui/index.html b/desktop/ui/index.html index 4a74b191615..7829f0d2970 100644 --- a/desktop/ui/index.html +++ b/desktop/ui/index.html @@ -18,7 +18,13 @@ #phases li[data-state="done"] { color: #4b8b3b; } #phases li[data-state="failed"] { color: #b3261e; font-weight: 600; } #failure { margin-top: 1.25rem; display: grid; gap: .75rem; } - #failure[hidden] { display: none; } + #failure[hidden], #consent[hidden] { display: none; } + #consent { margin-top: 1.25rem; display: grid; gap: .75rem; } + #consent p { margin: 0; } + #consentTarget { display: grid; grid-template-columns: max-content 1fr; gap: .25rem .75rem; margin: 0; } + #consentTarget dt { color: #666; } + #consentTarget dd { margin: 0; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: .85rem; } + #consentNote { color: #666; font-size: .85rem; } .actions { display: flex; gap: .6rem; align-items: center; } button { border: 0; border-radius: .5rem; padding: .6rem 1rem; background: #2563eb; color: white; cursor: pointer; font: inherit; } button.secondary { background: #e3e3e8; color: #202124; } @@ -32,6 +38,7 @@ #phases li[data-state="active"] { color: #f5f5f7; } #phases li[data-state="done"] { color: #8bd17c; } #phases li[data-state="failed"] { color: #ff8a80; } + #consentTarget dt, #consentNote { color: #bbb; } button.secondary { background: #3a3a3c; color: #f5f5f7; } textarea { background: #1c1c1e; border-color: #ffffff22; } } @@ -43,6 +50,19 @@

OpenCodex

Starting OpenCodex…

    +