From f764765c6453a718806d3465ea966015fa233123 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 13:25:41 +0900 Subject: [PATCH 01/75] fix: release 2.68.0 blockers and bring the menu bar patches to the Windows/Linux tray (#6052) --- desktop/src-tauri/src/native_tray.rs | 17 ++++- .../030_done.md | 35 +++++++++ devlog/_plan/260927_release_2680/000_plan.md | 32 ++++++++ .../010_wp4_blockers_and_tray.md | 47 ++++++++++++ .../260927_release_2680/020_wp5_release.md | 11 +++ gui/src/pages/Tray.tsx | 74 +++++++++++++++---- gui/src/pages/tray-data.ts | 49 +++++++++--- gui/src/pages/tray.css | 19 ++++- gui/tests/tray-data.test.ts | 34 ++++++++- src/client/link-relay.ts | 13 ++++ src/client/link-state.ts | 54 ++++++++++++++ src/client/runtime.ts | 14 +++- src/providers/kiro-model-catalog.ts | 19 +++-- structure/companion.md | 2 + structure/remote-link.md | 2 +- tests/clients/client-link-relay.test.ts | 20 +++++ tests/clients/client-link-state.test.ts | 36 ++++++++- .../providers/kiro/kiro-model-catalog.test.ts | 25 +++++++ .../responses-grok-devin-preflight.test.ts | 9 ++- 19 files changed, 476 insertions(+), 36 deletions(-) create mode 100644 devlog/_plan/260927_directive_marker_bridge/030_done.md create mode 100644 devlog/_plan/260927_release_2680/000_plan.md create mode 100644 devlog/_plan/260927_release_2680/010_wp4_blockers_and_tray.md create mode 100644 devlog/_plan/260927_release_2680/020_wp5_release.md diff --git a/desktop/src-tauri/src/native_tray.rs b/desktop/src-tauri/src/native_tray.rs index 58731cb04d3..583c3378314 100644 --- a/desktop/src-tauri/src/native_tray.rs +++ b/desktop/src-tauri/src/native_tray.rs @@ -334,11 +334,21 @@ fn publish(app: &AppHandle, generation: u64, binding: Option, sn *state .cache .lock() - .unwrap_or_else(std::sync::PoisonError::into_inner) = (binding, snapshot); + .unwrap_or_else(std::sync::PoisonError::into_inner) = (binding, cached(snapshot)); } }); } +/// The cached copy every later refresh starts from. `switchFailed` answers one switch, so it is +/// delivered once and never cached: carried forward, it would settle the next switch's spinner on +/// that switch's first loading publish. +fn cached(mut snapshot: Value) -> Value { + if let Some(fields) = snapshot.as_object_mut() { + fields.remove("switchFailed"); + } + snapshot +} + fn display_payload(snapshot: Value) -> Option<(Value, Vec)> { let bytes = serde_json::to_vec(&snapshot).ok()?; if bytes.len() <= 8 * 1024 * 1024 { @@ -406,6 +416,11 @@ mod tests { } #[test] fn refresh_failure_preserves_age_and_clears_busy_state() { + let reported = json!({"refreshing":false,"errors":["refused"],"switchFailed":true}); + let kept = cached(reported); + assert!(kept.get("switchFailed").is_none()); + assert_eq!(kept["errors"], json!(["refused"])); + let before = json!({"updatedAt":12,"refreshing":true,"today":{"totalTokens":30}}); let after = failed(before, "Unavailable"); assert_eq!(after["updatedAt"], 12); diff --git a/devlog/_plan/260927_directive_marker_bridge/030_done.md b/devlog/_plan/260927_directive_marker_bridge/030_done.md new file mode 100644 index 00000000000..103964137d8 --- /dev/null +++ b/devlog/_plan/260927_directive_marker_bridge/030_done.md @@ -0,0 +1,35 @@ +# 030 — done: Codex App visualization references for routed models + +## Outcome + +Any routed model can now show a Codex App inline visualization. #6040 (`a1285fc648`) stopped the +citation filter from deleting non-citation directives; #6045 (`dc784d3e6f`) rewrites the private-use +`visualize` reference into the app's own `::codex-inline-vis{…}` directive in the text routed models +read, so a model whose provider drops private-use characters still reads and writes a form the app renders. + +## Evidence + +- App bundle 26.924.22138: `f2`, `Err` and `i3e` (see 001) render the ASCII directive directly. +- Local app rebuilt from `dc784d3e6f` with a forced standalone sidecar; the installed `ocx` reports + 2.68.0 and contains `codex-inline-vis`; `codesign --verify --deep --strict` passes. +- Live, through the installed proxy: `anthropic/claude-opus-5-5` and `cursor/claude-opus-5-5` answered a + private-use reference with `::codex-inline-vis{path="/tmp/demo-chart.html"}`; `gpt-6-luna` returned the + private-use form unchanged; `xai/grok-4.7` received and quoted the private-use form. +- Render: a Claude-routed final answer carrying `::codex-inline-vis{…}` rendered as an interactive widget in + Codex App; the widget reported `route: Claude, version: #6045, outcome: renders` back to the thread. +- Reviews: architect Fermat, auditor Lovelace, reviewer Archimedes (132,715 `f2` parity cases), and two + release regression lanes over #6040 and #6045 found no regression. + +## What did not go to plan + +- `desktop/scripts/prepare-sidecar.ts` reuses `dist/standalone/*/ocx` when it exists, so the first rebuilt + app shipped a stale 2.61.0 proxy. The standalone build has to be forced before `prepare-sidecar`. +- A commentary message is recorded as a reasoning summary on this route, which the app does not render + as markdown directives; the render check needed a final answer. +- #6045 merged by admin at the owner's instruction before its PR CI finished; the post-merge lane=all run + on `dc784d3e6f` is the CI evidence for the merged tree. + +## Follow-ups + +- Make `prepare-sidecar` rebuild when the source is newer than the cached standalone binary. +- Replies already stored in the bare `visualize{…}` form are not repaired. diff --git a/devlog/_plan/260927_release_2680/000_plan.md b/devlog/_plan/260927_release_2680/000_plan.md new file mode 100644 index 00000000000..8cfde2f80a8 --- /dev/null +++ b/devlog/_plan/260927_release_2680/000_plan.md @@ -0,0 +1,32 @@ +# 260927 release 2.68.0 — plan + +## Reader summary + +The owner asked (2026-09-27) for a main..dev regression review with astra reviewers, for the +Windows/Linux tray to gain the menu bar patches the macOS panel received, and for a full 2.68.0 +release. `origin/main` is v2.67.0 (`4bc92294aa`); `origin/dev` (`dc784d3e6f`) carries 86 commits +beyond it. Procedure follows [the 2.67.0 round](../../_fin/260926_release_2670/020_wp3_release.md); +only values differ. The owner asked for CI to be judged heuristically: a failure is a blocker only +when it reproduces or is tied to a change in the range. + +## Work-phase map + +| wp | Doc | Change | +|---|---|---| +| wp4 | [010](010_wp4_blockers_and_tray.md) | release blockers from the review, Windows tray parity | +| wp5 | [020](020_wp5_release.md) | candidate CI, pre-move to 2.69.0, promotion, publish, verify | + +## Review lanes (astra, read-only) + +| Lane | Range | Verdict | +|---|---|---| +| #6040 citation filter | `a1285fc648` | OK (700,168 comparisons) | +| #6045 visualization references | `dc784d3e6f` | OK | +| dev sanity + CI classification | whole range | OK; Devin test mock leak (test-only) | +| merge trains 1-5 | #5901..#5909 | OK | +| desktop, batches 6-8 | #5910..#5957 | BLOCK: native tray `switchFailed` cached | +| Kiro series | #5967..#6016 | BLOCK: first discovery failure never backs off | +| batches 9-10 | #5984..#6031 | BLOCK: Home-initiated Remote Link answers 503 | + +Owner disposition for the Remote Link finding (2026-09-27): keep 2.67.0 behaviour for Home-initiated +links only; Child-initiated links keep the new ownership proof. diff --git a/devlog/_plan/260927_release_2680/010_wp4_blockers_and_tray.md b/devlog/_plan/260927_release_2680/010_wp4_blockers_and_tray.md new file mode 100644 index 00000000000..cae311c70e6 --- /dev/null +++ b/devlog/_plan/260927_release_2680/010_wp4_blockers_and_tray.md @@ -0,0 +1,47 @@ +# 010 — wp4: release blockers and Windows tray parity + +## Fixes (one PR to dev) + +| Finding | Files | Change | Proof | +|---|---|---|---| +| Kiro discovery failure without a last good list retried on every request | `src/providers/kiro-model-catalog.ts` | `failedUntil` map carries the 60 s retry per account identity; cleared on success and by `clearKiroAccountModels` | new case in `tests/providers/kiro/kiro-model-catalog.test.ts` fails before (2 calls), passes after (1) | +| Native tray `switchFailed` cached and settling later switches | `desktop/src-tauri/src/native_tray.rs` | `cached()` drops the one-shot flag from the snapshot every refresh starts from | `cargo test --lib native_tray` | +| Home-initiated Child relays answer 503 | `src/client/link-relay.ts`, `src/client/runtime.ts` | explicit `HOME_INITIATED_LINK_TUNNEL` gate for a link-mode runtime without a sidecar; a relay with no gate still refuses | new case in `tests/clients/client-link-relay.test.ts`; the lane's repro passes | +| Devin preflight test leaks its adapter mock | `tests/responses/responses-grok-devin-preflight.test.ts` | restore the module in `afterAll` | 3-file run 82 pass (was 55/27) | + +## Windows/Linux tray parity (web tray `gui/src/pages/Tray.tsx`) + +| macOS patch | Web tray before | Change | +|---|---|---| +| #5920 windows that report data | present | none | +| #5922 provider marks, severity bars | missing | `ProviderIcon` in headings; `quotaSeverity` classes on bars (70/90) | +| #5931 switch the active account | missing | `switchState`/`exhausted` in `parseAccounts`, `accountSwitchRequest` routes, "Use this account" on hover/focus, pending and failure states | +| #5921 widget reload | not applicable (macOS widget) | none | + +The popup sends the switch with the dashboard session (`window.fetch` is session-wrapped by +`gui/src/api.ts`), not the desktop capability the native panel uses. + +## Verification + +`bun run typecheck`, GUI `tsc` and lint, `bun run structure:check`, `bun run privacy:scan`, the focused +test files above, `gui/tests/tray-data.test.ts`, and a Playwright capture of the web tray against the +running proxy. Full suite: PR CI. + +## Audit round 1 (Carver, astra) — FAIL, dispositions + +1. GUI build: `providerSources` `flatMap` inferred only `'codex'` — folded (`flatMap`; `tsc -p tsconfig.app.json` exit 0). +2. A Child-initiated link whose sidecar was deleted took the Home-initiated gate — folded. A join now + writes `link/child-initiated.json` with the link id; the runtime uses the Home gate only when neither + the sidecar nor a matching marker exists, and an unreadable marker counts as present. The auditor's + repro now answers 503 with no send for deleted and corrupt sidecars and for hub transport. + Residual: a 2.67.0 join made before this marker existed, with its sidecar later deleted, is treated as + Home-initiated, which is the 2.67.0 behaviour for every link. +3. Switch settled before the reload — folded: the row stays pending until the reload the switch started completes. +4. Unbounded PUT — folded: `createBoundedFetch(20 s)`; a timeout reports the switch failure. +Note (not folded): the web tray does not print `blockedReason`; a blocked account simply offers no "Use". +5. Round 2: pre-marker joins — partly folded. The runtime records the marker at start whenever the + sidecar is intact (`recordChildInitiatedLink`), so a pre-marker join that starts once on 2.68.0 is + protected from then on. Rebutted for the remaining case, a pre-marker join whose sidecar was deleted + before its first 2.68.0 start: that state is indistinguishable from a 2.67.0 Home-initiated link, and + failing it closed would break every existing Home-initiated link, which the owner chose to keep working. + It keeps exactly the 2.67.0 behaviour, the owner-accepted baseline. diff --git a/devlog/_plan/260927_release_2680/020_wp5_release.md b/devlog/_plan/260927_release_2680/020_wp5_release.md new file mode 100644 index 00000000000..2a79bda6dd5 --- /dev/null +++ b/devlog/_plan/260927_release_2680/020_wp5_release.md @@ -0,0 +1,11 @@ +# 020 — wp5: release 2.68.0 + +Values for the 2.67.0 procedure: `CAND` = `origin/dev` after wp4 merges; `PV=2.68.0-preview.20260927`; +pre-move `dev-version-bump.yml --ref main -f intended-version=2.68.0 -f mode=pre-move` (dev → 2.69.0); +promotion branches `codex/260927-release-preview-2.68.0` and `codex/260927-release-main-2.68.0` built +with `git merge -s ours` and `scripts/release-version-sources.ts`; merge commits (never squash); push-event +CI and Service lifecycle at both promotion SHAs; `release.yml` preview first, then stable; verify npm +dist-tags, both GitHub releases' assets, and `latest.json` signatures; fast-forward local branches. + +Heuristic CI rule (owner): a failing job blocks only when it reproduces on rerun or its log points at a +change in main..dev. Runner-stall signatures get one job rerun. diff --git a/gui/src/pages/Tray.tsx b/gui/src/pages/Tray.tsx index af4c2bcbfa9..f3cc5b34ea3 100644 --- a/gui/src/pages/Tray.tsx +++ b/gui/src/pages/Tray.tsx @@ -1,13 +1,19 @@ -import { useEffect, useState } from 'react'; +import { useEffect, useRef, useState } from 'react'; +import { createBoundedFetch } from '../bounded-fetch'; import { useI18n } from '../i18n/shared'; import { formatTokens } from '../format-tokens'; import { formatProviderDisplayName } from '../provider-icons'; +import { ProviderIcon } from '../components/provider-workspace/ProviderRail'; +import { quotaSeverity } from '../quota-summary'; import { UsageCompanionChart } from './usage-companion-chart'; import { companionTimelineQuery, companionTimelineProjection, type CompanionSettings, type CompanionSettingsResponse, type UsageTimeline } from './usage-companion-utils'; -import { fetchTrayJson, parseTrayUsage, filterUsage, measuredTotals, finite, parseAccounts, providerSources, quotaWindows, relativeReset, type TrayProvider, type TrayTotals, type TrayUsage } from './tray-data'; +import { accountSwitchRequest, fetchTrayJson, parseTrayUsage, filterUsage, measuredTotals, finite, parseAccounts, providerSources, quotaWindows, relativeReset, type TrayProvider, type TraySwitchKind, type TrayTotals, type TrayUsage } from './tray-data'; declare global { interface Window { __OPENCODEX_TRAY_VISIBLE__?: boolean } } +/** A switch that has not answered by then is reported as failed rather than left spinning. */ +const SWITCH_TIMEOUT_MS = 20_000; + const incomplete = (data: TrayUsage | null | undefined) => data?.usageIncomplete || data?.historyTruncated || data?.entriesTruncated; export default function Tray() { @@ -23,7 +29,30 @@ export default function Tray() { const [updatedAt, setUpdatedAt] = useState(null); const [refreshing, setRefreshing] = useState(false); const [revision, setRevision] = useState(0); + const [switching, setSwitching] = useState(null); + const [switchError, setSwitchError] = useState(null); const retry = () => setRevision(value => value + 1); + // Set after a successful switch: the pending row stays busy until the reload it started lands. + const awaitingRefresh = useRef(false); + // The same route and body the native panel sends; the dashboard session supplies the + // credentials. The switch settles only when the reload shows the runtime's own selection. + const switchAccount = async (provider: string, kind: TraySwitchKind, accountId: string) => { + setSwitching(`${provider}:${accountId}`); + setSwitchError(null); + const bounded = createBoundedFetch(SWITCH_TIMEOUT_MS); + try { + const request = accountSwitchRequest(provider, kind, accountId); + const response = await fetch(request.path, { method: 'PUT', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(request.body), signal: bounded.signal }); + if (!response.ok) throw new Error(String(response.status)); + awaitingRefresh.current = true; + retry(); + } catch { + setSwitching(null); + setSwitchError(t(kind === 'codex' ? 'codexAuth.switchFailed' : 'prov.accountSwitchFail')); + } finally { + bounded.clear(); + } + }; // Every post-await state write checks both effect disposal and the request's AbortSignal. // react-doctor-disable-next-line react-doctor/no-set-state-after-await-in-effect @@ -49,7 +78,7 @@ export default function Tray() { try { const sources = providerSources(await json('/api/config')); const rows = await Promise.all(sources.map(async source => { - if (!source.path) return { name: source.name, accounts: [] }; + if (!source.path) return { name: source.name, switchKind: source.switchKind, accounts: [] }; try { const payload = await json>(source.path); if (source.name === 'openai') { @@ -58,9 +87,9 @@ export default function Tray() { payload.activeCodexAccountId = selection.activeCodexAccountId ?? '__main__'; } catch { /* Missing selection is unknown, never inferred from quota. */ } } - return { name: source.name, accounts: parseAccounts(payload) }; + return { name: source.name, switchKind: source.switchKind, accounts: parseAccounts(payload) }; } - catch { return { name: source.name, accounts: [], unavailable: true }; } + catch { return { name: source.name, switchKind: source.switchKind, accounts: [], unavailable: true }; } })); if (active()) { setProviders(rows); setQuotaError(false); hadSuccess = true; } } catch { if (active()) { setProviders([]); setQuotaError(true); } } @@ -95,7 +124,11 @@ export default function Tray() { await Promise.allSettled([quotas, metrics]); } finally { busy = false; - if (active()) { setRefreshing(false); if (hadSuccess) setUpdatedAt(Date.now()); } + if (active()) { + setRefreshing(false); + if (hadSuccess) setUpdatedAt(Date.now()); + if (awaitingRefresh.current) { awaitingRefresh.current = false; setSwitching(null); } + } if (!disposed && current.signal.aborted && visible()) void load(); } }; @@ -152,11 +185,23 @@ export default function Tray() { } {(settings?.showAccounts ?? true) &&
{quotaError &&

{t('startup.tray.unavailable')}

} + {switchError &&

{switchError}

} {providers.filter(provider => !hiddenProviders.has(provider.name)).map(provider =>
-

{formatProviderDisplayName(provider.name, t)}

+

{formatProviderDisplayName(provider.name, t)}

{!provider.accounts.length &&
{t(provider.unavailable ? 'startup.tray.unavailable' : 'pws.dashboard.noQuota')}
} - {provider.accounts.map(account =>
-
{account.label}{account.plan}{account.active && ●}
+ {provider.accounts.map(account => { + const kind = provider.switchKind; + const pending = switching === `${provider.name}:${account.id}`; + return
+
+ {account.label}{account.exhausted && ⚠} + + {pending && {t('pws.accountSwitching')}} + {!pending && kind && account.switchState === 'available' && } + {account.plan} + {account.active && ✓} + +
{account.email && account.email !== account.label &&
{account.email}
} {!quotaWindows(account.quota).length &&
{t(account.unavailable ? 'startup.tray.unavailable' : 'pws.dashboard.noQuota')}
} {quotaWindows(account.quota).map(window => { @@ -165,11 +210,12 @@ export default function Tray() { const percent = finite(window.percent) ? Math.min(100, window.percent) : null; return
{label}{percent === null ? '—' : `${Math.round(percent)}%`} - - -
; - })} -
)} + + +
; + })} +
; + })} )}
}
{t('tray.updated', { time: updatedAt === null ? '—' : new Date(updatedAt).toLocaleTimeString(locale, { hour: '2-digit', minute: '2-digit' }) })}
diff --git a/gui/src/pages/tray-data.ts b/gui/src/pages/tray-data.ts index ef58b46aec5..72fa5dc06c8 100644 --- a/gui/src/pages/tray-data.ts +++ b/gui/src/pages/tray-data.ts @@ -6,24 +6,54 @@ import { normalizeQuotaForPlan } from '../codex-quota-utils'; export type TrayTotals = Partial>; export type TrayModel = TrayTotals & { model: string; provider: string }; export interface TrayUsage { summary: TrayTotals; models: TrayModel[]; customWindow?: boolean; since?: number; until?: number; usageIncomplete?: boolean; historyTruncated?: boolean; entriesTruncated?: boolean } -export interface TrayAccount { unavailable?: boolean; id: string; label: string; quota: AccountQuota | null; plan?: string; active?: boolean; email?: string; status?: string; quotaFailure?: string } -export interface TrayProvider { name: string; accounts: TrayAccount[]; unavailable?: boolean } -export interface TrayProviderSource { name: string; path: string | null } +/** + * Whether this account can be made active from the tray. Mirrors the native panel + * (`desktop/src-tauri/src/native_tray_accounts.rs` `switch_state`): only what the switch route + * itself refuses is blocked, and an exhausted account stays switchable. + */ +export type TraySwitchState = 'active' | 'available' | 'blocked'; +export type TraySwitchKind = 'codex' | 'oauth' | 'apiKey'; +export interface TrayAccount { unavailable?: boolean; id: string; label: string; quota: AccountQuota | null; plan?: string; active?: boolean; email?: string; status?: string; quotaFailure?: string; switchState?: TraySwitchState; blockedReason?: 'mainHardLock' | 'paused' | 'validationPending'; exhausted?: boolean } +export interface TrayProvider { name: string; accounts: TrayAccount[]; unavailable?: boolean; switchKind?: TraySwitchKind | null } +export interface TrayProviderSource { name: string; path: string | null; switchKind: TraySwitchKind | null } export const finite = (value: unknown): value is number => typeof value === 'number' && Number.isFinite(value) && value >= 0; const object = (value: unknown): Record => value !== null && typeof value === 'object' && !Array.isArray(value) ? value as Record : {}; // Read only the safe management projection; never retain configuration credentials. export function providerSources(value: unknown): TrayProviderSource[] { - return Object.entries(object(object(value).providers)).flatMap(([name, raw]) => { + return Object.entries(object(object(value).providers)).flatMap(([name, raw]) => { const config = object(raw); if (config.disabled === true) return []; - const path = name === 'openai' ? '/api/codex-auth/accounts' - : config.authMode === 'oauth' ? `/api/oauth/accounts?${new URLSearchParams({ provider: name, quota: '1' })}` - : config.hasApiKey === true && config.authMode !== 'forward' ? `/api/providers/keys?${new URLSearchParams({ name, quota: '1' })}` : null; - return [{ name, path }]; + if (name === 'openai') return [{ name, path: '/api/codex-auth/accounts', switchKind: 'codex' as const }]; + if (config.authMode === 'oauth') return [{ name, path: `/api/oauth/accounts?${new URLSearchParams({ provider: name, quota: '1' })}`, switchKind: 'oauth' as const }]; + if (config.hasApiKey === true && config.authMode !== 'forward') return [{ name, path: `/api/providers/keys?${new URLSearchParams({ name, quota: '1' })}`, switchKind: 'apiKey' as const }]; + return [{ name, path: null, switchKind: null }]; }); } +/** The management request that makes `accountId` active; the same routes and bodies the native panel and dashboard use. */ +export function accountSwitchRequest(provider: string, kind: TraySwitchKind, accountId: string): { path: string; body: Record } { + if (kind === 'codex') return { path: '/api/codex-auth/active', body: { accountId } }; + if (kind === 'oauth') return { path: '/api/oauth/accounts/active', body: { provider, accountId } }; + return { path: '/api/providers/keys/active', body: { name: provider, id: accountId } }; +} + +function switchState(row: Record, active: boolean): Pick { + if (active) return { switchState: 'active' }; + if (object(row.mainAccountHardLock).state === 'blocked') return { switchState: 'blocked', blockedReason: 'mainHardLock' }; + if (row.paused === true) return { switchState: 'blocked', blockedReason: 'paused' }; + if (object(row.health).reason === 'validation_pending') return { switchState: 'blocked', blockedReason: 'validationPending' }; + return { switchState: 'available' }; +} + +/** Same windows as `isCodexQuotaExhausted`: 100% in a window the plan is governed by, or in the burst window. */ +function exhausted(quota: Record, plan: string | undefined): boolean { + const monthlyOnly = ['go', 'free'].includes((plan ?? '').trim().toLowerCase()); + const short = quota.fiveHourPercent ?? quota.shortPercent; + const windows = monthlyOnly ? [quota.monthlyPercent, short] : [quota.weeklyPercent, quota.monthlyPercent, short]; + return windows.some(value => typeof value === 'number' && Number.isFinite(value) && value >= 100); +} + export function parseAccounts(value: unknown): TrayAccount[] { const body = object(value); const rows = body.accounts ?? body.keys; @@ -36,7 +66,8 @@ export function parseAccounts(value: unknown): TrayAccount[] { const quota = row.quotaUnavailable === true || row.quotaMode === 'unsupported' || !row.quota ? null : object(row.quota) as unknown as AccountQuota; const plan = typeof row.plan === 'string' ? row.plan : undefined; const activeId = body.activeAccountId ?? body.activeId ?? body.activeCodexAccountId; - return { id: row.id, label, email, plan, unavailable: row.quotaUnavailable === true, active: typeof activeId === 'string' ? activeId === row.id : row.active === true, status: typeof object(row.health).status === 'string' ? object(row.health).status as string : undefined, quotaFailure: typeof row.quotaFailure === 'string' ? row.quotaFailure : undefined, quota: normalizeQuotaForPlan(quota, plan) }; + const active = typeof activeId === 'string' ? activeId === row.id : row.active === true; + return { id: row.id, label, email, plan, unavailable: row.quotaUnavailable === true, active, ...switchState(row, active), exhausted: quota !== null && exhausted(object(row.quota), plan), status: typeof object(row.health).status === 'string' ? object(row.health).status as string : undefined, quotaFailure: typeof row.quotaFailure === 'string' ? row.quotaFailure : undefined, quota: normalizeQuotaForPlan(quota, plan) }; }); } diff --git a/gui/src/pages/tray.css b/gui/src/pages/tray.css index 9b969780eb4..2b78372b10e 100644 --- a/gui/src/pages/tray.css +++ b/gui/src/pages/tray.css @@ -25,6 +25,8 @@ html.tray-document { --tray-fill: rgba(235, 235, 245, 0.11); --tray-fill-strong: rgba(235, 235, 245, 0.18); --tray-accent: #32d74b; + --tray-warn: #ffd60a; + --tray-critical: #ff453a; --tray-alert: #ff9f8f; --tray-radius: 12px; --tray-numerals: ui-rounded, 'SF Pro Rounded', -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif; @@ -147,7 +149,10 @@ html.tray-document[data-tray-vibrancy='on'] { } .tray-provider + .tray-provider { margin-top: 12px; } -.tray-provider h2 { margin-bottom: 4px; } +.tray-provider h2 { margin-bottom: 4px; display: flex; align-items: center; gap: 6px; } +/* The dashboard's own provider mark, shrunk to the heading; the native panel shows the same file. */ +.tray-provider-icon.provider-icon { width: 18px; height: 18px; border-radius: 4px; } +.tray-provider-icon img, .tray-provider-icon .provider-icon-mask { width: 12px; height: 12px; } .tray-account { padding-left: 10px; border-left: 1px solid var(--tray-separator); margin-top: 7px; } .tray-account-name { display: flex; @@ -161,6 +166,15 @@ html.tray-document[data-tray-vibrancy='on'] { margin-bottom: 4px; } .tray-account-meta { flex-shrink: 0; color: var(--tray-label-tertiary); font-size: 10px; } +.tray-account-label { min-width: 0; overflow: hidden; text-overflow: ellipsis; } +.tray-active { color: var(--tray-accent); } +.tray-exhausted { color: var(--tray-warn); } +/* "Use" appears on hover or keyboard focus, as in the native panel, and keeps its space so the + row never shifts; screens without hover always show it. */ +.tray-page .tray-use { opacity: 0; margin-right: 6px; padding: 0 5px; font-size: 10px; text-decoration: none; border: 1px solid var(--tray-separator); color: var(--tray-label-secondary); } +.tray-account:hover .tray-use, .tray-page .tray-use:focus-visible { opacity: 1; } +.tray-page .tray-use:disabled { cursor: default; } +@media (hover: none) { .tray-page .tray-use { opacity: 1; } } .tray-account-email { color: var(--tray-label-tertiary); font-size: 10px; margin: -2px 0 4px; } .tray-quota { display: grid; grid-template-columns: 94px 32px minmax(35px, 1fr) 78px; align-items: center; gap: 7px; margin-top: 4px; font-size: 10px; } @@ -169,6 +183,9 @@ html.tray-document[data-tray-vibrancy='on'] { .tray-quota time { text-align: right; white-space: nowrap; color: var(--tray-label-tertiary); overflow: hidden; text-overflow: ellipsis; } .tray-bar { height: 4px; background: var(--tray-fill-strong); overflow: hidden; border-radius: 2px; } .tray-bar i { display: block; height: 100%; background: var(--tray-accent); border-radius: inherit; } +/* The dashboard strip's thresholds: 70% warns, 90% is critical. */ +.tray-bar--warn i { background: var(--tray-warn); } +.tray-bar--critical i { background: var(--tray-critical); } .tray-missing { color: var(--tray-label-tertiary); font-size: 11px; } .tray-error { color: var(--tray-alert); font-size: 11px; } diff --git a/gui/tests/tray-data.test.ts b/gui/tests/tray-data.test.ts index e0d41a8e55b..b4392b0aa67 100644 --- a/gui/tests/tray-data.test.ts +++ b/gui/tests/tray-data.test.ts @@ -1,5 +1,5 @@ import { describe, expect, test } from 'bun:test'; -import { fetchTrayJson, parseTrayUsage, filterUsage, measuredTotals, parseAccounts, providerSources, quotaWindows, relativeReset, resetTimestamp } from '../src/pages/tray-data'; +import { accountSwitchRequest, fetchTrayJson, parseTrayUsage, filterUsage, measuredTotals, parseAccounts, providerSources, quotaWindows, relativeReset, resetTimestamp } from '../src/pages/tray-data'; import type { CompanionSettings } from '../src/pages/usage-companion-utils'; describe('tray data', () => { @@ -93,3 +93,35 @@ test('tray fetch works without AbortSignal static helpers and forwards cancellat Object.defineProperty(AbortSignal, 'timeout', timeout); } }); + +describe('tray account switching (parity with the native panel)', () => { + test('each source names the switch route the native panel uses, and forward sources have none', () => { + const sources = providerSources({ providers: { openai: {}, claude: { authMode: 'oauth' }, key: { hasApiKey: true }, fwd: { hasApiKey: true, authMode: 'forward' } } }); + expect(sources.map(s => [s.name, s.switchKind])).toEqual([['openai', 'codex'], ['claude', 'oauth'], ['key', 'apiKey'], ['fwd', null]]); + expect(accountSwitchRequest('openai', 'codex', 'a1')).toEqual({ path: '/api/codex-auth/active', body: { accountId: 'a1' } }); + expect(accountSwitchRequest('claude', 'oauth', 'a2')).toEqual({ path: '/api/oauth/accounts/active', body: { provider: 'claude', accountId: 'a2' } }); + expect(accountSwitchRequest('key', 'apiKey', 'k3')).toEqual({ path: '/api/providers/keys/active', body: { name: 'key', id: 'k3' } }); + }); + + test('only what the switch route refuses is blocked; an exhausted account stays switchable', () => { + const rows = parseAccounts({ activeAccountId: 'on', accounts: [ + { id: 'on', quota: { weeklyPercent: 10 } }, + { id: 'lock', mainAccountHardLock: { state: 'blocked' }, quota: { weeklyPercent: 98 } }, + { id: 'paused', paused: true, quota: { weeklyPercent: 1 } }, + { id: 'pending', health: { reason: 'validation_pending' }, quota: { weeklyPercent: 1 } }, + { id: 'spent', quota: { weeklyPercent: 100 } }, + { id: 'burst', plan: 'go', quota: { weeklyPercent: 100, fiveHourPercent: 40, monthlyPercent: 20 } }, + { id: 'short', quota: { shortPercent: 100, weeklyPercent: 20 } }, + ] }); + expect(rows.map(r => [r.id, r.switchState, r.blockedReason ?? null, r.exhausted])).toEqual([ + ['on', 'active', null, false], + ['lock', 'blocked', 'mainHardLock', false], + ['paused', 'blocked', 'paused', false], + ['pending', 'blocked', 'validationPending', false], + ['spent', 'available', null, true], + // A monthly-only plan is not governed by its weekly window. + ['burst', 'available', null, false], + ['short', 'available', null, true], + ]); + }); +}); diff --git a/src/client/link-relay.ts b/src/client/link-relay.ts index 351abefbd31..41fcd0f8eb2 100644 --- a/src/client/link-relay.ts +++ b/src/client/link-relay.ts @@ -39,6 +39,19 @@ export interface LinkTunnelGate { waitForConnected(timeoutMs: number, signal?: AbortSignal): Promise; } +/** + * The gate for a Home-initiated link. The Home runs `ssh -R` and owns the forward, so the Child + * has no tunnel supervisor and no local SSH process whose socket it could prove. This gate keeps + * the 2.67.0 behaviour for that link only: forward to the tunnel port without an ownership proof, + * never hold, and answer 503 when the connection is refused. Child-initiated links keep their + * supervisor, and a relay with no gate at all still refuses every request. + */ +export const HOME_INITIATED_LINK_TUNNEL: LinkTunnelGate = { + connected: () => true, + pending: () => false, + waitForConnected: async () => false, +}; + export interface LinkRelayDeps { fetchImpl?: typeof fetch; clock?: LinkRelayClock; diff --git a/src/client/link-state.ts b/src/client/link-state.ts index c81a1c0c951..180b79fc775 100644 --- a/src/client/link-state.ts +++ b/src/client/link-state.ts @@ -38,6 +38,52 @@ export function clientLinkStatePath(configDir?: string): string { return join(linkDir(configDir), "client-link.json"); } +/** + * Durable evidence that this machine joined its link itself. It survives a lost sidecar, so a + * Child-initiated link whose sidecar is gone fails closed instead of being mistaken for a + * Home-initiated one, which the relay forwards without a tunnel ownership proof. + */ +export function childLinkMarkerPath(configDir?: string): string { + return join(linkDir(configDir), "child-initiated.json"); +} + +/** True when the marker names `linkId`, or when a marker exists but cannot be read. */ +export function isChildInitiatedLink(linkId: string, path: string = childLinkMarkerPath()): boolean { + let text: string; + try { + text = readFileSync(path, "utf8"); + } catch (error) { + return !isMissingPathError(error); + } + try { + const raw = JSON.parse(text) as { linkId?: unknown }; + return typeof raw?.linkId !== "string" || raw.linkId === linkId; + } catch { + return true; + } +} + +/** + * Record the marker for a join made before markers existed. Called at runtime start while the + * sidecar is still intact, so a sidecar deleted afterwards cannot turn that link into a + * Home-initiated one. A missing or unreadable sidecar records nothing. + */ +export function recordChildInitiatedLink( + linkId: string, + path: string = clientLinkStatePath(), + marker: string = join(dirname(path), "child-initiated.json"), +): void { + let sidecar: ClientLinkState | null; + try { + sidecar = readClientLinkState(path); + } catch { + return; + } + if (!sidecar || sidecar.linkId !== linkId || isChildInitiatedLink(linkId, marker)) return; + atomicWriteFile(marker, `${JSON.stringify({ linkId })}\n`); + if (process.platform !== "win32") chmodSync(marker, 0o600); +} + export function parseClientLinkState(value: unknown): ClientLinkState { if (!value || typeof value !== "object" || Array.isArray(value)) throw new ClientLinkStateError("client-link.json is not an object"); const raw = value as Record; @@ -91,6 +137,9 @@ export function writeClientLinkState(state: ClientLinkState, path: string = clie // the explicit mode on the final path. atomicWriteFile(path, `${JSON.stringify(normalized, null, 2)}\n`); if (process.platform !== "win32") chmodSync(path, 0o600); + const marker = join(dir, "child-initiated.json"); + atomicWriteFile(marker, `${JSON.stringify({ linkId: normalized.linkId })}\n`); + if (process.platform !== "win32") chmodSync(marker, 0o600); } /** @@ -106,5 +155,10 @@ export function clearClientLinkState(expectedLinkId: string, path: string = clie if (isMissingPathError(error)) return false; throw error; } + try { + unlinkSync(join(dirname(path), "child-initiated.json")); + } catch (error) { + if (!isMissingPathError(error)) throw error; + } return true; } diff --git a/src/client/runtime.ts b/src/client/runtime.ts index 4667bb342a9..fa87ab20b4a 100644 --- a/src/client/runtime.ts +++ b/src/client/runtime.ts @@ -17,8 +17,9 @@ import { findAvailablePort, isAddrInUse, PortUnavailableError, waitForPortAvaila import type { ReplacementStartRequest } from "../server/restart-replacement"; import type { OcxClientConnectionConfig } from "../types"; import { createLinkKeySource } from "./link-ingress"; +import { HOME_INITIATED_LINK_TUNNEL } from "./link-relay"; import { createClientLinkSupervisor, type ClientLinkSupervisor } from "./link-tunnel"; -import { clientLinkStatePath } from "./link-state"; +import { clientLinkStatePath, isChildInitiatedLink, recordChildInitiatedLink } from "./link-state"; import { startMachineListener, type MachineListenerDeps } from "./machine-listener"; import { isLinkConnection, readClientConnectionState } from "./state"; @@ -278,12 +279,21 @@ export async function startClientRuntime( const preferred = linkMode ? config.port : options.port ?? config.port ?? 10100; // Share one cached key source between the listener and its tunnel supervisor. const linkKey = linkMode ? createLinkKeySource(state.value.tokenFingerprint) : undefined; + // Joins made before the marker existed get one now, while their sidecar is still here. + if (linkMode && state.value.link) { + try { recordChildInitiatedLink(state.value.link.linkId); } catch { /* the sidecar check below still applies */ } + } const supervisor = linkMode && existsSync(clientLinkStatePath()) ? createClientLinkSupervisor({ onLinkEnded: () => scheduleStandaloneRecycle(state.value.tokenFingerprint), linkKey, }) : null; + // A Child with no sidecar and no record of joining itself was connected by its Home over + // `ssh -R`; that link keeps 2.67.0's unproven forward. A Child-initiated link that lost its + // sidecar gets no gate at all, so the relay refuses it. + const homeInitiated = linkMode && !supervisor && !!state.value.link + && !isChildInitiatedLink(state.value.link.linkId); const { server, port: boundPort } = await bindClientListener({ state: state.value, linkMode, @@ -291,7 +301,7 @@ export async function startClientRuntime( explicitPort: options.port !== undefined, configuredPort: config.port, ...(linkMode ? { linkStatus: () => supervisor?.status() ?? { kind: "stopped" as const }, linkKeySource: linkKey } : {}), - ...(supervisor ? { linkTunnel: supervisor } : {}), + ...(supervisor ? { linkTunnel: supervisor } : homeInitiated ? { linkTunnel: HOME_INITIATED_LINK_TUNNEL } : {}), }, io); activeServer = server; activePort = boundPort; diff --git a/src/providers/kiro-model-catalog.ts b/src/providers/kiro-model-catalog.ts index 183c5f81f4f..2fb0d4ff2cc 100644 --- a/src/providers/kiro-model-catalog.ts +++ b/src/providers/kiro-model-catalog.ts @@ -19,6 +19,8 @@ interface Row { identity: string; models: KiroAccountModel[]; observedAt: number interface Flight { identity: string; promise: Promise } const rows = new Map(); const flights = new Map(); +/** Retry time after a failed discovery for an account that has no last good row to carry it. */ +const failedUntil = new Map(); export function kiroModelDiscoveryEnabled(): boolean { return process.env.OPENCODEX_KIRO_MODEL_DISCOVERY !== "0"; @@ -77,6 +79,8 @@ export function refreshKiroAccountModelsDetached( if (currentIdentity(account.id) !== identity) return; const old = validRow(account); if (old && Date.now() < old.nextRefreshAt) return; + const failed = failedUntil.get(account.id); + if (!old && failed?.identity === identity && Date.now() < failed.at) return; if (flights.get(account.id)?.identity === identity) return; const flight = (async (): Promise => { @@ -107,9 +111,14 @@ export function refreshKiroAccountModelsDetached( } if (currentIdentity(account.id) !== identity) return; const now = Date.now(); - if (fresh) rows.set(account.id, { identity, models: fresh, observedAt: now, - nextRefreshAt: now + KIRO_MODEL_CATALOG_TTL_MS }); - else if (old) rows.set(account.id, { ...old, nextRefreshAt: now + FAILURE_RETRY_MS }); + if (fresh) { + rows.set(account.id, { identity, models: fresh, observedAt: now, + nextRefreshAt: now + KIRO_MODEL_CATALOG_TTL_MS }); + failedUntil.delete(account.id); + } else if (old) rows.set(account.id, { ...old, nextRefreshAt: now + FAILURE_RETRY_MS }); + // Without a last good row the failure still has to back off, or every serving request + // after a restart would start another discovery while the endpoint is failing. + else failedUntil.set(account.id, { identity, at: now + FAILURE_RETRY_MS }); })(); flights.set(account.id, { identity, promise: flight }); void flight.catch(() => {}).finally(() => { @@ -143,8 +152,8 @@ export function kiroObservedContextWindow(model: string): number | undefined { } export function clearKiroAccountModels(accountId?: string): void { - if (accountId) { rows.delete(accountId); flights.delete(accountId); } - else { rows.clear(); flights.clear(); } + if (accountId) { rows.delete(accountId); flights.delete(accountId); failedUntil.delete(accountId); } + else { rows.clear(); flights.clear(); failedUntil.clear(); } } /** Deterministic test seam; production requests never call this. */ diff --git a/structure/companion.md b/structure/companion.md index 74e667c96ea..386a856edaa 100644 --- a/structure/companion.md +++ b/structure/companion.md @@ -77,6 +77,8 @@ and `exhausted` follows `isCodexQuotaExhausted` (100% in a governing window or t The "Use" action (`app/Sources/NativeTray/AccountSwitch.swift`) appears on hover, keyboard focus and as an accessibility action; an exhausted account stays switchable with a warning. +The web tray (`gui/src/pages/Tray.tsx`, the Windows and Linux popup) shows the same account state. `gui/src/pages/tray-data.ts` mirrors the native projection: `providerSources` names each provider's switch kind from the same config rules, `parseAccounts` derives `switchState`, `blockedReason` and `exhausted` with the same rules, and `accountSwitchRequest` builds the same route and body. The popup sends it with the dashboard session instead of the desktop capability, then reloads. Provider headings use the dashboard's `ProviderIcon`, and bars use `quotaSeverity` (warn 70%, critical 90%). `gui/tests/tray-data.test.ts` pins the parity. + The Tauri title reads `usage_today()`, matching the widget and retained Swift client. Every refresh applies the resulting optional title so icon-only clears an old counter. A nonblank custom template takes precedence over icon-only; unavailable measurements render as an em dash, not as a request diff --git a/structure/remote-link.md b/structure/remote-link.md index 81d52007e08..18bf22612e9 100644 --- a/structure/remote-link.md +++ b/structure/remote-link.md @@ -44,7 +44,7 @@ Applying a link probes the host key into a temporary file, waits for the operato ## Client link transport -The Child relay requires a positive `connected()` verdict from `src/client/link-tunnel.ts` before every fetch. Before the first keyed readiness probe and after a tunnel restart, the supervisor asynchronously proves the exact `127.0.0.1:` LISTEN socket belongs to its SSH child; an adopted process also needs matching pidfile argv and start time. `src/server/port-reclaim.ts` uses Linux `/proc/net/tcp{,6}` plus the expected PID's `/proc//fd` socket symlinks without external tools, macOS `lsof` and Windows `netstat` with bounded asynchronous execution. Every key-bearing probe and relayed request makes a fresh bounded asynchronous owner lookup; an adopted process also has its current argv and start time rechecked on every admission. An unreadable or timed-out identity denies only that admission and is retried; a confirmed mismatch releases the adopted PID without signalling it and lets the next tick start a fresh tunnel. An unknown socket-owner lookup likewise denies the current admission. A missing supervisor, or a failed or stopped tunnel, returns a retryable 503 without sending the link key or body to the persisted loopback port. A held request rechecks the verdict after its wait and before each retry. +The Child relay requires a positive `connected()` verdict from `src/client/link-tunnel.ts` before every fetch. Before the first keyed readiness probe and after a tunnel restart, the supervisor asynchronously proves the exact `127.0.0.1:` LISTEN socket belongs to its SSH child; an adopted process also needs matching pidfile argv and start time. `src/server/port-reclaim.ts` uses Linux `/proc/net/tcp{,6}` plus the expected PID's `/proc//fd` socket symlinks without external tools, macOS `lsof` and Windows `netstat` with bounded asynchronous execution. Every key-bearing probe and relayed request makes a fresh bounded asynchronous owner lookup; an adopted process also has its current argv and start time rechecked on every admission. An unreadable or timed-out identity denies only that admission and is retried; a confirmed mismatch releases the adopted PID without signalling it and lets the next tick start a fresh tunnel. An unknown socket-owner lookup likewise denies the current admission. A missing supervisor, or a failed or stopped tunnel, returns a retryable 503 without sending the link key or body to the persisted loopback port. A Home-initiated Child is the exception: the Home owns that link's `ssh -R` forward, so the Child has no sidecar, no supervisor and no SSH process whose socket it could prove. A join writes `/link/child-initiated.json` (the link id, mode 0600) beside the sidecar and removes it with the sidecar, so a Child-initiated link that lost its sidecar is recognised (`isChildInitiatedLink` in `src/client/link-state.ts`; an unreadable marker counts as present) and gets no gate. Only a link-mode runtime with neither the sidecar nor a marker for its link id is treated as Home-initiated: `src/client/runtime.ts` passes it `HOME_INITIATED_LINK_TUNNEL` (`src/client/link-relay.ts`), which keeps the 2.67.0 behaviour for that link only: forward without an ownership proof, never hold, and answer 503 at once when the connection is refused. The race recorded below therefore still applies in full to Home-initiated links. A held request rechecks the verdict after its wait and before each retry. The listener can change between an ownership check and the TCP connect. This is a pre-existing race class: since #5801/#5818 the Child relay has sent the link key to 127.0.0.1: without an ownership check. Future hardening on Unix can forward through a socket in a private mode-0700 directory (ssh -L /sock:...), removing the competing TCP listener from that path. diff --git a/tests/clients/client-link-relay.test.ts b/tests/clients/client-link-relay.test.ts index 2aae2d60503..a8cd6fce202 100644 --- a/tests/clients/client-link-relay.test.ts +++ b/tests/clients/client-link-relay.test.ts @@ -5,6 +5,7 @@ import { join } from "node:path"; import type { Server } from "bun"; import { forwardLinkRequestHeaders, + HOME_INITIATED_LINK_TUNNEL, LINK_RELAY_BODY_MAX_BYTES, LINK_RELAY_HEADER_TIMEOUT_MS, LINK_RELAY_HOLD_MS, @@ -474,6 +475,25 @@ describe("client link relay while the tunnel reconnects", () => { }); } + test("a Home-initiated link forwards like 2.67.0 and never holds a refused request", async () => { + // The Home owns the `ssh -R` forward, so this Child has no supervisor to prove it. The explicit + // Home-initiated gate forwards (a missing gate above still refuses); a refused connection is + // answered at once with a retryable 503 instead of waiting for a reconnect nobody drives. + let sends = 0; + const ok = await relayLinkDataRequestImpl(relayRequest({ method: "POST", body: "{}" }), target, { + tunnel: HOME_INITIATED_LINK_TUNNEL, + fetchImpl: (async () => { sends += 1; return Response.json({ forwarded: true }); }) as typeof fetch, + }); + expect(ok.status).toBe(200); + expect(sends).toBe(1); + const refused = await relayLinkDataRequestImpl(relayRequest({ method: "POST", body: "{}" }), target, { + tunnel: HOME_INITIATED_LINK_TUNNEL, + fetchImpl: (async () => { sends += 1; throw new Error("connection refused"); }) as typeof fetch, + }); + expect(refused.status).toBe(503); + expect(sends).toBe(2); + }); + test("rechecks connected state before retrying a refused fetch", async () => { let connected = true; let pending = false; diff --git a/tests/clients/client-link-state.test.ts b/tests/clients/client-link-state.test.ts index 17cada029b9..874564319d9 100644 --- a/tests/clients/client-link-state.test.ts +++ b/tests/clients/client-link-state.test.ts @@ -1,8 +1,11 @@ import { afterEach, expect, test } from "bun:test"; -import { existsSync, statSync, writeFileSync } from "node:fs"; +import { existsSync, statSync, unlinkSync, writeFileSync } from "node:fs"; import { createTempHome, type TempHome } from "../helpers/temp-home"; import { + childLinkMarkerPath, clearClientLinkState, + isChildInitiatedLink, + recordChildInitiatedLink, clientLinkStatePath, ClientLinkStateError, readClientLinkState, @@ -66,3 +69,34 @@ test("client link sidecar clear is owner checked", () => { expect(existsSync(path)).toBe(false); expect(clearClientLinkState(fixture().linkId, path)).toBe(false); }); + +test("a join leaves a marker that outlives a lost sidecar and is removed with it", () => { + // Without the marker a Child-initiated link whose sidecar went missing would look + // Home-initiated, and the relay would forward it without a tunnel ownership proof. + home = createTempHome("ocx-client-link-marker-"); + const path = clientLinkStatePath(home.configDir); + const marker = childLinkMarkerPath(home.configDir); + expect(isChildInitiatedLink("lnk_0123456789abcdef", marker)).toBe(false); + writeClientLinkState(fixture(), path); + expect(isChildInitiatedLink("lnk_0123456789abcdef", marker)).toBe(true); + expect(isChildInitiatedLink("lnk_fedcba9876543210", marker)).toBe(false); + if (process.platform !== "win32") expect(statSync(marker).mode & 0o777).toBe(0o600); + expect(clearClientLinkState("lnk_0123456789abcdef", path)).toBe(true); + expect(existsSync(marker)).toBe(false); + // An unreadable marker fails closed. + writeFileSync(marker, "{not json"); + expect(isChildInitiatedLink("lnk_fedcba9876543210", marker)).toBe(true); +}); + +test("a join made before markers existed gets one at start while its sidecar is intact", () => { + home = createTempHome("ocx-client-link-legacy-"); + const path = clientLinkStatePath(home.configDir); + const marker = childLinkMarkerPath(home.configDir); + writeClientLinkState(fixture(), path); + unlinkSync(marker); // what a 2.67.0 join left behind + recordChildInitiatedLink("lnk_fedcba9876543210", path, marker); + expect(existsSync(marker)).toBe(false); // another link's id records nothing + recordChildInitiatedLink("lnk_0123456789abcdef", path, marker); + unlinkSync(path); + expect(isChildInitiatedLink("lnk_0123456789abcdef", marker)).toBe(true); +}); diff --git a/tests/providers/kiro/kiro-model-catalog.test.ts b/tests/providers/kiro/kiro-model-catalog.test.ts index 6dcd554d31e..19892a3c899 100644 --- a/tests/providers/kiro/kiro-model-catalog.test.ts +++ b/tests/providers/kiro/kiro-model-catalog.test.ts @@ -265,3 +265,28 @@ test("the roster adds at most 64 observed ids to the catalog", async () => { expect(result.models).toHaveLength(1 + 64); }); +test("a first discovery failure backs off even with no last good list", async () => { + // After a restart there is no cached row to carry the retry time, so a failing endpoint + // must still be tried once per retry window rather than once per serving request. + setup(); + const account = await add("fresh"); + let calls = 0; + const failing = { + resolveAddresses: async (url: string) => ({ hostname: new URL(url).hostname, + addresses: [{ address: "1.1.1.1", family: 4 }], privateNetwork: false }), + pinnedPost: async () => { calls++; return new Response("unavailable", { status: 503 }); }, + } as never; + refreshKiroAccountModelsDetached(account, provider, failing); + await awaitKiroModelRefreshForTests(account.id); + await Bun.sleep(1); // let the finished flight leave the join table + refreshKiroAccountModelsDetached(account, provider, failing); + await awaitKiroModelRefreshForTests(account.id); + await Bun.sleep(1); // let the finished flight leave the join table + expect(calls).toBe(1); + expect(readKiroAccountModels(account)).toBeUndefined(); + clearKiroAccountModels(account.id); + refreshKiroAccountModelsDetached(account, provider, failing); + await awaitKiroModelRefreshForTests(account.id); + await Bun.sleep(1); // let the finished flight leave the join table + expect(calls).toBe(2); +}); diff --git a/tests/responses/responses-grok-devin-preflight.test.ts b/tests/responses/responses-grok-devin-preflight.test.ts index 0647cfbd4c2..78d3bc5740d 100644 --- a/tests/responses/responses-grok-devin-preflight.test.ts +++ b/tests/responses/responses-grok-devin-preflight.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, expect, mock, test } from "bun:test"; +import { afterAll, afterEach, beforeEach, expect, mock, test } from "bun:test"; import type { ProviderAdapter } from "../../src/adapters/base"; import type { AdapterEvent, OcxConfig, OcxProviderConfig } from "../../src/types"; import { saveCredential } from "../../src/oauth/store"; @@ -9,6 +9,8 @@ import { createTempHome } from "../helpers/temp-home"; const resolver = await import("../../src/server/adapter-resolve"); const originalResolve = resolver.resolveAdapter; +// A snapshot, not the live namespace: mock.module rewrites that namespace in place. +const originalResolverModule = { ...resolver }; let events: AdapterEvent[] = []; let calls = 0; let blockedRun: ProviderAdapter["runTurn"]; @@ -46,6 +48,11 @@ beforeEach(async () => { expires: Date.now() + 3_600_000, accountId: "fixture", }); }); +// Bun keeps a module mock for the rest of the process. Without this, every later file in a +// non-isolated run resolves Devin through the synthetic rate-limited adapter above. +afterAll(() => { + mock.module("../../src/server/adapter-resolve", () => originalResolverModule); +}); afterEach(() => { try { release?.(); From 99d0a9400ebd8dedbefee9882698ecfb1f7bbaa7 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Sun, 27 Sep 2026 13:29:08 +0900 Subject: [PATCH 02/75] chore(release): open dev at 2.69.0 before releasing 2.68.0 (#6053) Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- desktop/src-tauri/Cargo.lock | 2 +- desktop/src-tauri/Cargo.toml | 2 +- desktop/src-tauri/tauri.conf.json | 2 +- package.json | 2 +- 4 files changed, 4 insertions(+), 4 deletions(-) diff --git a/desktop/src-tauri/Cargo.lock b/desktop/src-tauri/Cargo.lock index 1b024ffb041..320a9a2b840 100644 --- a/desktop/src-tauri/Cargo.lock +++ b/desktop/src-tauri/Cargo.lock @@ -2551,7 +2551,7 @@ dependencies = [ [[package]] name = "opencodex-desktop" -version = "2.68.0" +version = "2.69.0" dependencies = [ "base64 0.22.1", "dbus", diff --git a/desktop/src-tauri/Cargo.toml b/desktop/src-tauri/Cargo.toml index 89a5642fe99..ed8f57bf1bb 100644 --- a/desktop/src-tauri/Cargo.toml +++ b/desktop/src-tauri/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "opencodex-desktop" -version = "2.68.0" +version = "2.69.0" description = "OpenCodex desktop shell" authors = ["OpenCodex contributors"] license = "MIT" diff --git a/desktop/src-tauri/tauri.conf.json b/desktop/src-tauri/tauri.conf.json index 1bd4bbee3c9..a3790fcbc8c 100644 --- a/desktop/src-tauri/tauri.conf.json +++ b/desktop/src-tauri/tauri.conf.json @@ -1,7 +1,7 @@ { "$schema": "https://schema.tauri.app/config/2", "productName": "OpenCodex", - "version": "2.68.0", + "version": "2.69.0", "identifier": "com.opencodex.desktop", "build": { "frontendDist": "../ui", diff --git a/package.json b/package.json index 6785bca379d..82f92b4f3cc 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@bitkyc08/opencodex", - "version": "2.68.0", + "version": "2.69.0", "description": "Universal provider proxy for OpenAI Codex & Claude Code — use any LLM with Codex CLI/App/SDK and Claude Code", "type": "module", "main": "./bin/package-main.mjs", From 5a5edf9d5bd8f32f2228bf56be1d5badc65e76df Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:21:41 +0900 Subject: [PATCH 03/75] docs(devlog): plan merge train round 3 batch 1 --- .../_plan/260927_merge_train_3/000_roadmap.md | 30 +++++++++++++++++++ .../_plan/260927_merge_train_3/010_batch1.md | 30 +++++++++++++++++++ 2 files changed, 60 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/000_roadmap.md create mode 100644 devlog/_plan/260927_merge_train_3/010_batch1.md diff --git a/devlog/_plan/260927_merge_train_3/000_roadmap.md b/devlog/_plan/260927_merge_train_3/000_roadmap.md new file mode 100644 index 00000000000..a4a56c65274 --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/000_roadmap.md @@ -0,0 +1,30 @@ +# Merge train round 3 — roadmap + +Inventory at `dev` `99d0a9400e` (2026-09-27, after round 2 in `devlog/_plan/260927_merge_train_2/`). Round 2's +outcome carries forward: its owner-decision list (#5831, #5964, #5956, #5995, #5800, #5912, #5879) and its blocked list +(#5977 unauthenticated status proof, #5893 Bun NO_PROXY patterns, #5927 security, #5925 needs a split, #5953 overbroad +override, #5539, #5497, #5947, #5782, #4222 and the stale feature drafts) stay out of this lane unless their authors +changed the premise. + +Goal: land the open bug fixes and non-GUI enhancements that opened after round 2's inventory, close what they resolve, +and bring open issues and PRs to at most 40 each (61 issues and 80 PRs at the start). One lane, serialized batches, +each rebased on the newest `dev`. + +Every carried PR is one squashed commit with the author or a `Co-authored-by` trailer. Each item's GitHub page is read +through Aside's signed-in browser (captures in `.tmp/aside/`, gitignored) before it is carried or closed. Kimi +subagents review each PR and audit each batch diff; auth, credential and link-relay changes get a dedicated security +review whose specifics stay in scratch. + +## Batches + +| Batch | PRs | Issues | +|---|---|---| +| B1 | #6041, #6019, #6015, #6011, #6006, #6026, #6034 (security review) | #6033, #6017, #6014, #4191 (only if the fix, not just a pin, lands), #6005, #5960, #6032 | +| B2 | luvs01 non-GUI bug fixes: #6048, #6047, #6046, #6038, #6036, #6035, #6057; #6022 (mdwsk88) | per PR | +| B3 | #6020 or #6056 (same quota-activation code; pick one), #6027, rebased #6050 #6049 #6037, #6042 (security review), #6030, #6003 | #6018, #5569 | +| B4 | bugs found through Aside that have no PR yet, implemented in this lane | per issue | + +## Out of scope + +GUI changes: #6043, #6007, #5983, #5905, #5950, #6025, #6010, and the GUI feature drafts. #6044 is a conflicting draft +that also edits a GUI test. diff --git a/devlog/_plan/260927_merge_train_3/010_batch1.md b/devlog/_plan/260927_merge_train_3/010_batch1.md new file mode 100644 index 00000000000..0a4ca773a5c --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/010_batch1.md @@ -0,0 +1,30 @@ +# B1 — small bug fixes with owner-filed issues + +Base: `dev` `99d0a9400e`. Branch `codex/train3-b1`. + +| PR | Author | Issue | Change | Risk | +|---|---|---|---|---| +| #6041 | Ingwannu | #6033 | `abort_restart()` restores the pre-update wanted intent instead of forcing `wanted = true` (desktop/src-tauri exit.rs, updater.rs) | Tauri; hosted macOS/Windows/Linux desktop jobs | +| #6019 | Ingwannu | #6017 | macOS ACL parser stops trusting the `user:0`/`root` display name as root identity | plugin trust; must only tighten | +| #6015 | Ingwannu | #6014 | picker route test binds real listeners instead of probing then releasing ports | test only plus a runtime option | +| #6011 | Ingwannu | #4191 | pins established-WebSocket failure behavior with a test and ADR | issue closes only if behavior is fixed | +| #6006 | Ingwannu | #6005 | translated Anthropic output schemas claim `strict` only when strict-eligible | adapter contract | +| #6026 | codingbooo | #5960 | `ocx models` derives catalog price estimates when no manual price is set | CLI output | +| #6034 | Ingwannu | #6032 | link relay strips provider credential headers | security boundary; dedicated review | + +## Method + +1. Kimi review per PR (verdict LAND / LAND-WITH-FIXES / HOLD); #6034 also gets the security verdict. +2. Squash each PR onto the branch in the table order, preserving the author; fold review fixes as separate commits. +3. Reconcile test-layout registries and the file-size ratchet once for the batch. +4. Local: `bun install`, `bun run typecheck`, the focused test files each PR touches, `bun run structure:check`, + `bun run privacy:scan`. +5. Push, open the batch PR with the template, wait for exact-head CI, merge with `--admin --merge + --match-head-commit`, then close the source PRs and issues with the merge commit. + +## Aside evidence + +Captured to `.tmp/aside/` for every PR and issue above. All seven issues are owner-filed today with reproduction and +code pointers that match the PR premises. #4191's thread records the owner's position (Sep 21) that an established +WebSocket dying mid-turn is a failed leg rather than an SSE fallback, plus two contributor data sets (Sep 23) asking for +a fallback, so a test-and-ADR PR does not by itself resolve that issue. From 1c9f1fb015c56ead44d58161dd309e5ea8b36a01 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 14:21:42 +0900 Subject: [PATCH 04/75] test(claude): bind picker proxies without port probes (#6015) Carried from #6015 into merge train round 3. Co-authored-by: Ingwannu --- src/claude/intercept/runtime.ts | 7 +++- structure/clients/claude-desktop.md | 4 ++ .../claude-desktop-picker-routes.test.ts | 41 +++++++++---------- 3 files changed, 29 insertions(+), 23 deletions(-) diff --git a/src/claude/intercept/runtime.ts b/src/claude/intercept/runtime.ts index 82cd133ea16..4ca539ec070 100644 --- a/src/claude/intercept/runtime.ts +++ b/src/claude/intercept/runtime.ts @@ -117,6 +117,8 @@ export interface StartClaudeInterceptOptions { loadPickerRoutes?: () => Promise; /** Test seam: builds the picker runtime. */ createPicker?: (options: CreatePickerRuntimeOptions) => PickerRuntime; + /** Test seam: bind real CONNECT handlers on kernel-assigned ports without probe-and-release races. */ + startProxy?: typeof startConnectProxy; /** Test seams: the macOS `security` runner and platform for the picker runtime and controller. */ pickerSecurity?: SecurityRunner; pickerPlatform?: NodeJS.Platform; @@ -132,6 +134,7 @@ export async function startClaudeIntercept(options: StartClaudeInterceptOptio if (options.requestedPort === 0 && !explicitPort) return null; const configDir = options.configDir ?? getConfigDir(); const ca = await ensureLocalInterceptCaForStartup(configDir); + const startProxy = options.startProxy ?? startConnectProxy; const authToken = ensureClaudeInterceptProxyToken(configDir); const leaf = issueLocalInterceptLeaf(ca, CLAUDE_INTERCEPT_HOSTS); // Refresh an env we already own (e.g. a pre-auth proxy URL left by an upgrade) before the @@ -156,7 +159,7 @@ export async function startClaudeIntercept(options: StartClaudeInterceptOptio }); let proxy: ConnectProxyHandle; try { - proxy = await startConnectProxy(claudeInterceptProxyPort(options.config, options.publicPort), { + proxy = await startProxy(claudeInterceptProxyPort(options.config, options.publicPort), { interceptPort: listener.port!, // A real apply may recreate a missing token while this listener remains live. // Read current validated authority per CONNECT; absent/invalid means deny, not mint. @@ -187,7 +190,7 @@ export async function startClaudeIntercept(options: StartClaudeInterceptOptio const runtime = picker; const interceptPort = listener.port!; try { - pickerProxy = await startConnectProxy(claudePickerProxyPort(options.config, options.publicPort), { + pickerProxy = await startProxy(claudePickerProxyPort(options.config, options.publicPort), { interceptPort, // No authToken: Desktop's egressProxyUrl cannot present proxy credentials, so this // listener stays an unauthenticated loopback relay until the profile format can carry diff --git a/structure/clients/claude-desktop.md b/structure/clients/claude-desktop.md index 6f038c5dab5..d8dc2451c43 100644 --- a/structure/clients/claude-desktop.md +++ b/structure/clients/claude-desktop.md @@ -152,6 +152,10 @@ Code, trusting only the intercept CA) gets the `api.anthropic.com` intercept and blind, never the picker; a tunnel with Chromium's `Mozilla/` User-Agent (the app, trusting only the login keychain) is asked of the picker runtime (`src/claude/intercept/picker-runtime.ts`), which blind-tunnels every target except `claude.ai:443`. +Production always uses the configured adjacent ports. Lifecycle tests inject only the CONNECT +factory and bind the real handlers on kernel-assigned ports; this preserves request handling while +avoiding the false reservation created by probing and closing a port pair before the ephemeral TLS +listener starts. The injected factory does not change production port selection. The User-Agent is a routing hint, not a trust boundary: a client that fakes it reaches only what any local process already reaches (the `api.anthropic.com` intercept is on the Claude Code proxy too; the `claude.ai` relay verifies upstream and adds no credential) and breaks only its own TLS, diff --git a/tests/claude-integration/claude-desktop-picker-routes.test.ts b/tests/claude-integration/claude-desktop-picker-routes.test.ts index 9111785e46c..2e395d53c43 100644 --- a/tests/claude-integration/claude-desktop-picker-routes.test.ts +++ b/tests/claude-integration/claude-desktop-picker-routes.test.ts @@ -1,9 +1,9 @@ import { afterEach, beforeEach, describe, expect, test } from "bun:test"; import { mkdtempSync, readFileSync, writeFileSync } from "node:fs"; -import { createServer } from "node:net"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { applyDesktopPickerProfile, inspectDesktopPickerProfile } from "../../src/claude/desktop-picker-profile"; +import { startConnectProxy } from "../../src/claude/intercept/connect-proxy"; import { pickerCaCertPath, pickerCaFingerprints } from "../../src/claude/intercept/picker-ca"; import type { PickerListenerOptions } from "../../src/claude/intercept/picker-listener"; import { createPickerRuntime } from "../../src/claude/intercept/picker-runtime"; @@ -21,6 +21,7 @@ let handle: ClaudeInterceptHandle | null = null; const previous: Record = {}; const ENV_KEYS = ["OPENCODEX_HOME", "OPENCODEX_CLAUDE_DESKTOP_CONFIG_DIR", "CLAUDE_CONFIG_DIR"] as const; const LISTENER_PORT = 45_679; +const REQUESTED_PROXY_PORT = 45_600; const INTERCEPT = { kind: "intercept", port: LISTENER_PORT }; const BLIND = { kind: "blind" }; @@ -49,28 +50,12 @@ const security: SecurityRunner = async args => { } }; -async function canBind(port: number): Promise { - return new Promise(resolve => { - const server = createServer(); - server.once("error", () => resolve(false)); - server.listen({ port, host: "127.0.0.1", exclusive: true }, () => server.close(() => resolve(true))); - }); -} - -async function freePortPair(): Promise { - for (let attempt = 0; attempt < 50; attempt += 1) { - const port = 20_000 + Math.floor(Math.random() * 30_000); - if (await canBind(port) && await canBind(port + 1)) return port; - } - throw new Error("no free port pair"); -} - /** Start the intercept pair with picker mode wired, as the server lifecycle does. */ async function startPicker(saved: OcxConfig, onDispatch?: (req: Request) => Response): Promise { writeFileSync(join(root, "config.json"), JSON.stringify(saved)); - const port = await freePortPair(); + const requestedProxyPorts: number[] = []; handle = await startClaudeIntercept({ - config: config({ claudeCode: { intercept: { port } } }), + config: config({ claudeCode: { intercept: { port: REQUESTED_PROXY_PORT } } }), publicPort: 10100, configDir: root, dispatch: async req => onDispatch?.(req) ?? new Response("unused"), @@ -78,6 +63,13 @@ async function startPicker(saved: OcxConfig, onDispatch?: (req: Request) => Resp loadPickerRoutes: async () => ({ nativeSlugs: [], routedModels: [{ provider: "xai", id: "grok-4.7", contextWindow: 256_000 }] }), pickerSecurity: security, pickerPlatform: "darwin", + // A probe that closes before startup does not reserve anything: the lifecycle's own + // ephemeral TLS listener or another process can take the observed pair. Bind the real proxy + // handlers directly on port 0 so the kernel owns both allocations until teardown. + startProxy: async (requestedPort, proxyOptions) => { + requestedProxyPorts.push(requestedPort); + return startConnectProxy(0, proxyOptions); + }, createPicker: options => createPickerRuntime({ ...options, startListener: (async (_: PickerListenerOptions) => ({ port: LISTENER_PORT, close: async () => {} })) as never, @@ -85,7 +77,14 @@ async function startPicker(saved: OcxConfig, onDispatch?: (req: Request) => Resp refreshIntervalMs: 3_600_000, }), }); - return port; + if (!handle || handle.pickerProxyPort === null || !getClaudePickerRuntime()) { + throw new Error("picker fixture did not start every runtime component"); + } + expect(requestedProxyPorts).toEqual([REQUESTED_PROXY_PORT, REQUESTED_PROXY_PORT + 1]); + const boundPorts = [handle.listener.port, handle.proxyPort, handle.pickerProxyPort]; + expect(boundPorts.every(port => typeof port === "number" && Number.isInteger(port) && port > 0)).toBe(true); + expect(new Set(boundPorts).size).toBe(boundPorts.length); + return handle.pickerProxyPort; } async function dispatch(path: string, init: RequestInit = {}, deps: Parameters[3] = {}) { @@ -146,7 +145,7 @@ describe("first-party turns picker mode on by default", () => { expect(applied.status).toBe(200); expect(applied.body.picker).toMatchObject({ effective: true, reason: "restart_required", trust: "trusted", profile: "applied" }); expect(decision()).toEqual(INTERCEPT); - expect(inspectDesktopPickerProfile()).toMatchObject({ kind: "applied", proxyUrl: `http://127.0.0.1:${port + 1}` }); + expect(inspectDesktopPickerProfile()).toMatchObject({ kind: "applied", proxyUrl: `http://127.0.0.1:${port}` }); expect(keychain.trusted).toBe(true); expect(persisted().claudeCode?.intercept?.picker).toBeUndefined(); From 5747a5c4b9dbe44faef7cbd2213d7abe9a21fe35 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 14:21:44 +0900 Subject: [PATCH 05/75] fix(link): strip provider credentials at relay (#6034) Carried from #6034 into merge train round 3. Co-authored-by: Ingwannu --- .../src/content/docs/guides/remote-link.md | 2 +- src/client/link-relay.ts | 5 ++- ...ADR-6032-link-relay-credential-boundary.md | 12 ++++++ structure/remote-link.md | 4 +- tests/clients/client-link-relay.test.ts | 38 ++++++++++++++++--- 5 files changed, 53 insertions(+), 8 deletions(-) create mode 100644 structure/decisions/ADR-6032-link-relay-credential-boundary.md diff --git a/docs-site/src/content/docs/guides/remote-link.md b/docs-site/src/content/docs/guides/remote-link.md index 38037482146..cf5973216fb 100644 --- a/docs-site/src/content/docs/guides/remote-link.md +++ b/docs-site/src/content/docs/guides/remote-link.md @@ -73,7 +73,7 @@ When a step fails, the dashboard shows the reason and, when SSH reported one, th ## Security -The Child uses the Home computer's providers and provider credentials through the link. The Home creates a separate link key for each Child; removing the link revokes that key. On the Child, the key stays inside OpenCodex: credentials that Codex or Claude Code send there are not forwarded to the Home, and any program on the Child that reaches `127.0.0.1:` uses the Home without a key, the same local trust a standalone install gives. Web pages from other sites are refused. Compare the host fingerprint before confirmation so a wrong machine or changed host key is not accepted by mistake. Dashboard sessions issued from a Tailscale identity cannot manage machine links. +The Child uses the Home computer's providers and provider credentials through the link. The Home creates a separate link key for each Child; removing the link revokes that key. On the Child, the key stays inside OpenCodex: credentials that Codex or Claude Code send there are not forwarded to the Home, including Bearer, Azure `api-key`, Anthropic-compatible `x-api-key`, and Google `x-goog-api-key` forms. Any program on the Child that reaches `127.0.0.1:` uses the Home without a key, the same local trust a standalone install gives. Web pages from other sites are refused. Compare the host fingerprint before confirmation so a wrong machine or changed host key is not accepted by mistake. Dashboard sessions issued from a Tailscale identity cannot manage machine links. ## CLI reference diff --git a/src/client/link-relay.ts b/src/client/link-relay.ts index 41fcd0f8eb2..d2a155631b8 100644 --- a/src/client/link-relay.ts +++ b/src/client/link-relay.ts @@ -93,7 +93,10 @@ const RESPONSE_OMITTED_HEADERS = new Set(["content-encoding", "content-length"]) * Caller credentials never cross the tunnel. The Child's own ChatGPT or Anthropic credential * stays on the Child, and the Home sees exactly one admission: the link key. */ -const CALLER_CREDENTIAL_HEADERS = ["authorization", "x-api-key", "x-opencodex-api-key", "chatgpt-account-id", "cookie"] as const; +const CALLER_CREDENTIAL_HEADERS = [ + "authorization", "api-key", "x-api-key", "x-goog-api-key", "x-opencodex-api-key", + "chatgpt-account-id", "cookie", +] as const; function jsonError(status: number, error: string, retry = false): Response { const headers = retry ? { "Retry-After": String(LINK_RELAY_RETRY_AFTER_SECONDS) } : undefined; diff --git a/structure/decisions/ADR-6032-link-relay-credential-boundary.md b/structure/decisions/ADR-6032-link-relay-credential-boundary.md new file mode 100644 index 00000000000..d7861d0d8ae --- /dev/null +++ b/structure/decisions/ADR-6032-link-relay-credential-boundary.md @@ -0,0 +1,12 @@ +# ADR-6032 — Child link relay credential boundary + +- Contract owner: [Remote Link](../remote-link.md) + +## Decision record + +- Purpose and intent: Ensure a Child sends only its link admission key across the machine tunnel, never a caller's provider credential. +- Existing implementation and constraints: The relay already removed Bearer, `x-api-key`, OpenCodex, account and cookie credentials before attaching the link key. Built-in Azure and Google adapters use the independent `api-key` and `x-goog-api-key` forms, which were not in that denylist and therefore survived ordinary end-to-end header forwarding. +- Alternatives considered: Strip every header containing `key` or `token`; reuse the broad log-redaction regex; extend the relay's explicit list with the provider credential forms it actually supports. +- Chosen approach: Add `api-key` and `x-goog-api-key` to the case-insensitive explicit relay denylist and exercise them through both header construction and the actual fetch boundary. +- Why this approach: A broad name heuristic could remove legitimate protocol headers such as `idempotency-key`. The explicit list closes the proven built-in adapter paths while preserving ordinary request metadata and the existing link-key wire contract. +- Benefits, costs and impact: Azure and Google caller keys remain on the Child, matching the existing Anthropic/OpenAI behavior. Custom credential header names still require deliberate review before they become supported provider authentication forms. diff --git a/structure/remote-link.md b/structure/remote-link.md index 18bf22612e9..e4edc2419e0 100644 --- a/structure/remote-link.md +++ b/structure/remote-link.md @@ -54,6 +54,8 @@ Codex keeps the standalone loopback routing: `routingTarget` in `src/client/conn `src/client/link-ingress.ts` is the link-mode data plane of the machine listener. A `/v1/responses` WebSocket upgrade answers `426 upgrade_required`, which codex-rs maps to its HTTP fallback, and no upgrade is ever relayed. Every relayed route first passes the standalone loopback Host and Origin gate (`isAllowedRequestOrigin` in `src/server/auth-cors.ts`), so a rebinding or cross-site page gets `403 origin_rejected` and nothing is fetched upstream. `/readyz` is answered locally. The link key is read from the service token file once, when the listener starts, and held in memory; the file must hold the key whose fingerprint the connection committed. While it does not, relayed routes answer `503 link_credential_unavailable` without an upstream fetch, and the file is read again at most once a second; a valid key is never re-read. The key is never logged or returned. -`src/client/link-relay.ts` forwards exactly the `linkRouteAllowed` routes from `src/link/routes.ts` through the tunnel. It drops the caller's `Authorization`, `x-api-key`, `x-opencodex-api-key`, `chatgpt-account-id` and `cookie` and sends the link key as `Authorization: Bearer`, the wire an `env_key` config sent; `GET /v1/usage` takes it as `x-opencodex-api-key`, the only header that route admits. The Home admits the key and serves the Child with its own accounts. The request body is streamed chunk by chunk with the caller's `Content-Length` and a byte-counting cap at the inbound limit (`resolveInboundBodyLimitBytes`, 256 MiB by default); a larger declared or streamed body answers 413. A lone `Transfer-Encoding: chunked` without `Content-Length` is admitted as a standalone admits it, because the listener has already de-chunked the body; any other Transfer-Encoding, or one next to a `Content-Length`, answers 400. The Home's response headers may take up to 300 seconds, and a caller abort ends the wait sooner. SSE passes through chunk by chunk with caller-abort propagation and a 300-second idle limit, other response bodies stream under the same byte cap, and the relay answers 503 with Retry-After while the tunnel is down. The client supervisor is the relay's tunnel gate (`LinkTunnelGate`): only while the tunnel is connecting or reconnecting (including the start of the client runtime) does a relayed request wait, for at most 15 seconds (`LINK_RELAY_HOLD_MS`) from its first wait and with at most 64 requests waiting, before it is forwarded once; a connected tunnel costs one `pending()` call per request, and a failed one answers 503 at once. A forward whose connection was refused sent nothing, so while the tunnel reconnects it may wait again and be sent again inside the same 15 seconds, provided the streamed body was never read or cancelled; any other failure (a reset, a timeout, a failure after the body started) is never replayed. Both the Child's machine listener and the Home's hub-link listener bind with `idleTimeout: 255`, the public listener's limit, so a held or slow turn is not cut by Bun's 10-second default. Like a standalone data route, a relayed request then lifts its own idle timer (`server.timeout(req, 0)` in `src/client/link-ingress.ts`), so a quiet stretch longer than 255 seconds inside a long generation is not cut either; the relay's header deadline, SSE idle limit and caller abort bound the wait instead. Hub transport keeps the 4 MiB management-relay listener bound and its default idle limit. Link mode waits for the configured port without signalling its holder, then binds there or fails; `src/client/runtime.ts` passes the cached link key, tunnel status and tunnel gate through `bindClientListener` to every bind attempt. Link mode turns the management relay off and refuses key rotation and revocation, which belong to the hub. +`src/client/link-relay.ts` forwards exactly the `linkRouteAllowed` routes from `src/link/routes.ts` through the tunnel. It drops the caller's `Authorization`, Azure `api-key`, Anthropic-compatible `x-api-key`, Google `x-goog-api-key`, `x-opencodex-api-key`, `chatgpt-account-id` and `cookie` and sends the link key as `Authorization: Bearer`, the wire an `env_key` config sent; `GET /v1/usage` takes it as `x-opencodex-api-key`, the only header that route admits. The explicit provider forms matter because the Child and Home are separate credential owners: caller provider credentials never cross the tunnel merely because they are not Bearer tokens. The Home admits the link key and serves the Child with its own accounts. The request body is streamed chunk by chunk with the caller's `Content-Length` and a byte-counting cap at the inbound limit (`resolveInboundBodyLimitBytes`, 256 MiB by default); a larger declared or streamed body answers 413. A lone `Transfer-Encoding: chunked` without `Content-Length` is admitted as a standalone admits it, because the listener has already de-chunked the body; any other Transfer-Encoding, or one next to a `Content-Length`, answers 400. The Home's response headers may take up to 300 seconds, and a caller abort ends the wait sooner. SSE passes through chunk by chunk with caller-abort propagation and a 300-second idle limit, other response bodies stream under the same byte cap, and the relay answers 503 with Retry-After while the tunnel is down. The client supervisor is the relay's tunnel gate (`LinkTunnelGate`): only while the tunnel is connecting or reconnecting (including the start of the client runtime) does a relayed request wait, for at most 15 seconds (`LINK_RELAY_HOLD_MS`) from its first wait and with at most 64 requests waiting, before it is forwarded once; a connected tunnel costs one `pending()` call per request, and a failed one answers 503 at once. A forward whose connection was refused sent nothing, so while the tunnel reconnects it may wait again and be sent again inside the same 15 seconds, provided the streamed body was never read or cancelled; any other failure (a reset, a timeout, a failure after the body started) is never replayed. Both the Child's machine listener and the Home's hub-link listener bind with `idleTimeout: 255`, the public listener's limit, so a held or slow turn is not cut by Bun's 10-second default. Like a standalone data route, a relayed request then lifts its own idle timer (`server.timeout(req, 0)` in `src/client/link-ingress.ts`), so a quiet stretch longer than 255 seconds inside a long generation is not cut either; the relay's header deadline, SSE idle limit and caller abort bound the wait instead. Hub transport keeps the 4 MiB management-relay listener bound and its default idle limit. Link mode waits for the configured port without signalling its holder, then binds there or fails; `src/client/runtime.ts` passes the cached link key, tunnel status and tunnel gate through `bindClientListener` to every bind attempt. Link mode turns the management relay off and refuses key rotation and revocation, which belong to the hub. + +> Decision record: [ADR-6032](decisions/ADR-6032-link-relay-credential-boundary.md) Regression coverage lives in `tests/clients/link-ssh-argv.test.ts`, `tests/clients/link-ssh-config.test.ts`, `tests/clients/link-tunnel-state.test.ts`, `tests/clients/link-store.test.ts`, `tests/clients/link-boundary.test.ts`, `tests/clients/link-routes.test.ts`, `tests/clients/client-link-connect.test.ts`, `tests/clients/client-link-relay.test.ts`, `tests/clients/client-machine-listener.test.ts`, `tests/clients/client-link-status.test.ts`, `tests/clients/client-link-runtime.test.ts`, `tests/codex-integration/injection-link-websocket.test.ts`, `tests/clients/link-supervisor.test.ts`, `tests/clients/link-status-projection.test.ts`, `tests/clients/link-admission-wait.test.ts`, `tests/clients/link-fingerprint.test.ts`, `tests/cli/cli-link.test.ts`, `tests/server/link-management-routes.test.ts`, `tests/server/link-join-route.test.ts`, `tests/server/link-listener-lifecycle.test.ts`, `tests/clients/client-link-teardown.test.ts` and `gui/tests/remote-link.test.tsx`. diff --git a/tests/clients/client-link-relay.test.ts b/tests/clients/client-link-relay.test.ts index a8cd6fce202..01bd7c0969f 100644 --- a/tests/clients/client-link-relay.test.ts +++ b/tests/clients/client-link-relay.test.ts @@ -106,10 +106,13 @@ describe("client link HTTP relay", () => { test("replaces every caller credential with the link key and filters hop-by-hop headers", () => { const caller = new Headers({ Authorization: "Bearer caller-chatgpt-oauth", + "Api-Key": "azure-caller", "X-OpenCodex-API-Key": "ocx_data_caller", "X-Api-Key": "sk-ant-caller", + "X-Goog-Api-Key": "google-caller", "ChatGPT-Account-Id": "acct-caller", Cookie: "session=caller", + "Idempotency-Key": "idem-1", "X-Trace": "trace-1", Connection: "keep-alive, X-Remove", "X-Remove": "secret", @@ -118,15 +121,18 @@ describe("client link HTTP relay", () => { }); const forwarded = forwardLinkRequestHeaders(caller, LINK_KEY, "/v1/responses"); expect(forwarded.get("authorization")).toBe(`Bearer ${LINK_KEY}`); + expect(forwarded.get("idempotency-key")).toBe("idem-1"); expect(forwarded.get("x-trace")).toBe("trace-1"); - for (const name of ["x-opencodex-api-key", "x-api-key", "chatgpt-account-id", "cookie", "connection", "keep-alive", "x-remove", "host", "content-length"]) { + for (const name of ["api-key", "x-opencodex-api-key", "x-api-key", "x-goog-api-key", "chatgpt-account-id", "cookie", "connection", "keep-alive", "x-remove", "host", "content-length"]) { expect(forwarded.get(name)).toBeNull(); } // /v1/usage admits only the dedicated header on the Home. const usage = forwardLinkRequestHeaders(caller, LINK_KEY, "/v1/usage"); expect(usage.get("x-opencodex-api-key")).toBe(LINK_KEY); expect(usage.get("authorization")).toBeNull(); + expect(usage.get("api-key")).toBeNull(); expect(usage.get("x-api-key")).toBeNull(); + expect(usage.get("x-goog-api-key")).toBeNull(); const response = sanitizeLinkResponseHeaders(new Headers({ Connection: "X-Response-Secret", @@ -146,18 +152,32 @@ describe("client link HTTP relay", () => { const fetchImpl = (async (_input, init) => { sent.push(new Headers(init?.headers)); return Response.json({ ok: true }); }) as typeof fetch; const response = await relayLinkDataRequest(relayRequest({ method: "POST", - headers: { Authorization: "Bearer caller-chatgpt-oauth", "ChatGPT-Account-Id": "acct-caller", "Content-Type": "application/json" }, + headers: { + Authorization: "Bearer caller-chatgpt-oauth", + "Api-Key": "azure-caller", + "X-Goog-Api-Key": "google-caller", + "ChatGPT-Account-Id": "acct-caller", + "Content-Type": "application/json", + }, body: "{}", }), target, { fetchImpl }); expect(response.status).toBe(200); expect(sent[0]?.get("authorization")).toBe(`Bearer ${LINK_KEY}`); + expect(sent[0]?.get("api-key")).toBeNull(); + expect(sent[0]?.get("x-goog-api-key")).toBeNull(); expect(sent[0]?.get("chatgpt-account-id")).toBeNull(); const usage = await relayLinkDataRequest(new Request("http://127.0.0.1:10100/v1/usage", { - headers: { Authorization: "Bearer caller-chatgpt-oauth" }, + headers: { + Authorization: "Bearer caller-chatgpt-oauth", + "Api-Key": "azure-caller", + "X-Goog-Api-Key": "google-caller", + }, }), target, { fetchImpl }); expect(usage.status).toBe(200); expect(sent[1]?.get("x-opencodex-api-key")).toBe(LINK_KEY); expect(sent[1]?.get("authorization")).toBeNull(); + expect(sent[1]?.get("api-key")).toBeNull(); + expect(sent[1]?.get("x-goog-api-key")).toBeNull(); }); test("rejects TE/CL ambiguity and oversized requests before outbound I/O", async () => { @@ -387,6 +407,8 @@ describe("client link HTTP relay", () => { path: new URL(req.url).pathname + new URL(req.url).search, host: req.headers.get("host"), authorization: req.headers.get("authorization"), + apiKey: req.headers.get("api-key"), + googleApiKey: req.headers.get("x-goog-api-key"), dedicated: req.headers.get("x-opencodex-api-key"), contentLength: req.headers.get("content-length"), transferEncoding: req.headers.get("transfer-encoding"), @@ -401,14 +423,20 @@ describe("client link HTTP relay", () => { const body = JSON.stringify({ input: "hello" }); const response = await fetch(new URL("/v1/responses?trace=1", machine.url), { method: "POST", - headers: { "Content-Type": "application/json", Authorization: "Bearer caller-chatgpt-oauth", "X-OpenCodex-API-Key": "ocx_data_caller" }, + headers: { + "Content-Type": "application/json", + Authorization: "Bearer caller-chatgpt-oauth", + "Api-Key": "azure-caller", + "X-Goog-Api-Key": "google-caller", + "X-OpenCodex-API-Key": "ocx_data_caller", + }, body, }); expect(response.status).toBe(200); expect(await response.json()).toEqual({ relayed: true }); expect(received).toEqual({ method: "POST", path: "/v1/responses?trace=1", host: `127.0.0.1:${hub.port}`, - authorization: `Bearer ${LINK_KEY}`, dedicated: null, + authorization: `Bearer ${LINK_KEY}`, apiKey: null, googleApiKey: null, dedicated: null, contentLength: String(body.length), transferEncoding: null, body, }); expect((await fetch(new URL("/v1/unknown", machine.url))).status).toBe(404); From 5f4784c21cda95277b5b0f72025680be2369e0ff Mon Sep 17 00:00:00 2001 From: codingbo Date: Sun, 27 Sep 2026 14:21:45 +0900 Subject: [PATCH 06/75] fix(cli): derive catalog price estimates for models without manual overrides (#6026) Carried from #6026 into merge train round 3. Co-authored-by: codingbo --- .../docs/reference/cli/providers-accounts.md | 12 +++++++-- src/cli/models-runtime.ts | 9 +++++-- src/cli/models.ts | 7 ++++- structure/runtime.md | 2 +- tests/cli/cli-models-price.test.ts | 27 +++++++++++++++++-- tests/cli/cli-models.test.ts | 23 ++++++++++++++++ 6 files changed, 72 insertions(+), 8 deletions(-) diff --git a/docs-site/src/content/docs/reference/cli/providers-accounts.md b/docs-site/src/content/docs/reference/cli/providers-accounts.md index e344104d868..9805136a375 100644 --- a/docs-site/src/content/docs/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/reference/cli/providers-accounts.md @@ -657,6 +657,14 @@ catalog entries; `enable`, `disable`, and `provider` control visibility; `select provider allowlist; `context` controls provider context caps; and `shadow` manages background shadow-call interception. +Model prices are estimates in USD per million tokens. `ocx models --json` includes a +`price` object with `cost4` rates and their source; `ocx models price --json` keeps +`cost` for the saved override and reports resolved rates in `effectiveCost`. +Manual prices (including zero) take precedence, followed by the shared catalog and +verified official-price fallbacks. Unknown models return `null`; no price is invented. +Automatic defaults are derived on read and do not populate `modelCosts` in your config, +so catalog updates remain effective. Use `set-price` to save provider-specific rates. + Every per-model operation the dashboard offers is available here, so a headless install never needs the GUI to manage a catalog. `add`, `remove`, and `list-custom` work against the config file and apply to a running proxy through a catalog sync; the rest talk to the live management API and require the @@ -664,9 +672,9 @@ proxy to be running (`ocx start`, or an installed service). | Subcommand | Supported flags | Action | | --- | --- | --- | -| `list` (default) | `--provider `, `--json` | List models seeded in configured providers. | +| `list` (default) | `--provider `, `--json` | List models seeded in configured providers, with estimated input/output prices. | | `live` | `--provider `, `--json` | Read the running catalog, including models discovered at runtime. Rows are flagged `native`/`routed`, `custom`, and `enabled`/`disabled`. | -| `price ` | `--json` | Read the model's saved manual price override; no override means automatic pricing. | +| `price ` | `--json` | Read the saved manual override and effective price, including automatic catalog defaults. | | `set-price ` | `--input `, `--output `, `--cache-read `, `--cache-write `, `--auto`, `--json` | Set display prices in USD per 1M tokens. Input/output are required when setting; omitted cache rates become zero. `--auto` removes only this model's override. | | `add ` | `--display-name `, `--context-window `, `--modalities ` | Register a model the provider catalog does not advertise. | | `edit ` | `--model-id `, `--display-name `, `--context-window `, `--modalities `, `--json` | Edit a custom model. `-` clears a field; `0` clears the context window. | diff --git a/src/cli/models-runtime.ts b/src/cli/models-runtime.ts index 1be92b6c301..ad753f3002f 100644 --- a/src/cli/models-runtime.ts +++ b/src/cli/models-runtime.ts @@ -18,6 +18,7 @@ import { isValidProviderName } from "../config/provider-name"; import { isValidModelDiscoveryModelId } from "../providers/model-discovery-limits"; import { redactSecretString } from "../lib/redact"; import type { ProviderCostOverlay } from "../types"; +import { resolveMatchedPrice } from "../usage/cost"; import { MAX_COST4_RATE } from "../usage/expected-prices"; import { isValidCost4Rate } from "../usage/user-cost-overlays"; @@ -124,8 +125,12 @@ async function priceRequest(write: boolean, argv: string[], deps: RuntimeApiDeps if (!validPriceCost(stored)) throw new Error("Invalid model price response"); cost = { ...stored }; } - printData({ provider, modelId, cost }, wantsJson, [ - cost === null ? `${selector}: automatic pricing` : `${selector}: ${JSON.stringify(cost)} USD per 1M tokens`, + // The API map owns manual overrides; bundled defaults remain derived rather + // than being persisted as overrides that would mask later catalog updates. + const effectiveCost = cost ?? resolveMatchedPrice(provider, modelId, undefined, [])?.cost4 ?? null; + printData({ provider, modelId, cost, effectiveCost }, wantsJson, [ + effectiveCost === null ? `${selector}: automatic pricing (unknown)` + : `${selector}: ${JSON.stringify(effectiveCost)} USD per 1M tokens${cost === null ? " (automatic estimate)" : ""}`, ]); return; } diff --git a/src/cli/models.ts b/src/cli/models.ts index 0f917964085..791dbf70dc2 100644 --- a/src/cli/models.ts +++ b/src/cli/models.ts @@ -1,6 +1,7 @@ /** * `ocx models` subcommand — list configured models and manage custom models. */ +import { resolveMatchedPrice, type MatchedPrice } from "../usage/cost"; import { randomUUID } from "node:crypto"; import { createInterface } from "node:readline/promises"; import { syncModelsToCodex } from "../codex/sync"; @@ -85,6 +86,7 @@ interface ModelEntry { contextWindow: number | null; inputModalities: string[] | null; reasoningEfforts: string[] | null; + price: MatchedPrice | null; } /** @@ -132,6 +134,7 @@ function collectModels(config: OcxConfig, providerFilter?: string): ModelEntry[] contextWindow: configuredContextWindow(prov, model) ?? null, inputModalities: modalities, reasoningEfforts: efforts, + price: resolveMatchedPrice(provName, model), }); }; @@ -428,7 +431,9 @@ function handleConfiguredModels(args: string[]): void { for (const m of provModels) { const marker = m.isDefault ? " *" : ""; const ctx = m.contextWindow ? ` (${Math.round(m.contextWindow / 1000)}k)` : ""; - console.log(` ${m.model}${marker}${ctx}`); + const rates = m.price?.cost4; + const pricing = rates ? ` ~$${rates.input}/$${rates.output} input/output per 1M tokens` : " price unknown"; + console.log(` ${m.model}${marker}${ctx}${pricing}`); } console.log(); } diff --git a/structure/runtime.md b/structure/runtime.md index 3abe289fc4a..693f9926b41 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -26,7 +26,7 @@ Native steering follows [the shared WebSocket contract](transports/streaming-hea Responses admission and finalization are composed through the [core module ownership](transports/responses.md#core-module-ownership). Kiro's optional account-load admission is process-local and request-owned; its slot ends with the response body or cancellation. Other providers retain their admission path. -Catalog HTTP acquisition follows the [proxy-routing contract](catalog.md#remote-catalog-http-proxy-routing). +Catalog HTTP acquisition follows the [proxy-routing contract](catalog.md#remote-catalog-http-proxy-routing). CLI model lists expose shared catalog estimates as `price`; `models price` retains the saved override in `cost` and adds `effectiveCost`. Explicit zero overrides win; unknown prices remain null, and automatic defaults are never persisted to `modelCosts`. Covered by `tests/cli/cli-models.test.ts` and `tests/cli/cli-models-price.test.ts`. OAuth refresh coordination follows the [refresh-lock identity contract](catalog.md#accounts-namespaces-and-pool-rotation): a fresh unreadable lock remains held, and release requires matching descriptor identity. A failed path-identity probe preserves the refresh callback outcome. Cooperating lock metadata changes serialize through the existing SQLite mutation transaction; release keeps the descriptor open through identity comparison and any unlink, then closes it. Failed metadata writes remove only a matching owned path after successful coordination; unknown identity, failed probes or unavailable coordination retain the path for stale recovery. Async refresh work holds no metadata transaction. diff --git a/tests/cli/cli-models-price.test.ts b/tests/cli/cli-models-price.test.ts index 9adda799770..50aa79c07a0 100644 --- a/tests/cli/cli-models-price.test.ts +++ b/tests/cli/cli-models-price.test.ts @@ -45,19 +45,42 @@ describe("models manual price commands", () => { }); expect(result.code).toBe(0); expect(result.calls).toEqual([{ path: "/api/providers/custom-price/model-costs", method: "GET", body: undefined }]); - expect(JSON.parse(result.stdout)).toEqual({ provider: "custom-price", modelId: "org/model--fast", cost: COST }); + expect(JSON.parse(result.stdout)).toEqual({ provider: "custom-price", modelId: "org/model--fast", cost: COST, effectiveCost: COST }); }); test("missing own keys read as automatic, including prototype-shaped selectors", async () => { for (const modelId of ["missing", "__proto__", "constructor", "toString"]) { const result = await invoke("price", [`custom-price/${modelId}`, "--json"], { provider: "custom-price", modelCosts: {} }); expect(result.code).toBe(0); - expect(JSON.parse(result.stdout)).toEqual({ provider: "custom-price", modelId, cost: null }); + expect(JSON.parse(result.stdout)).toEqual({ provider: "custom-price", modelId, cost: null, effectiveCost: null }); } const automatic = await invoke("price", ["custom-price/missing"], { provider: "custom-price", modelCosts: {} }); expect(automatic.stdout).toContain("automatic pricing"); }); + test("automatic pricing exposes known vendor rates without creating an override", async () => { + const result = await invoke("price", ["custom-price/claude-sonnet-4-6", "--json"], { + provider: "custom-price", modelCosts: {}, + }); + expect(result.code).toBe(0); + const row = JSON.parse(result.stdout); + expect(row.cost).toBeNull(); + expect(row.effectiveCost).toEqual({ input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }); + expect(result.calls).toHaveLength(1); + }); + + test("explicit zero prices override known automatic rates", async () => { + const zero = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; + const result = await invoke("price", ["custom-price/claude-sonnet-4-6", "--json"], { + provider: "custom-price", modelCosts: { "claude-sonnet-4-6": zero }, + }); + expect(JSON.parse(result.stdout)).toMatchObject({ cost: zero, effectiveCost: zero }); + const automatic = await invoke("price", ["custom-price/claude-sonnet-4-6"], { + provider: "custom-price", modelCosts: {}, + }); + expect(automatic.stdout).toContain("USD per 1M tokens (automatic estimate)"); + }); + test("set-price sends four numeric rates with omitted cache rates defaulted to zero", async () => { const result = await invoke("set-price", ["custom-price/org/model", "--input", "1.25", "--output", "5", "--json"]); expect(result.code).toBe(0); diff --git a/tests/cli/cli-models.test.ts b/tests/cli/cli-models.test.ts index 14c426c97d1..fcc99236780 100644 --- a/tests/cli/cli-models.test.ts +++ b/tests/cli/cli-models.test.ts @@ -60,6 +60,29 @@ describe("ocx models", () => { await warmModuleGraph({ graph: "cli-index/models", entry: cliPath }); }, COLD_SPAWN_WARMUP_HOOK_BUDGET_MS); + test("configured models expose automatic prices, overrides and unknowns without persisting defaults", () => { + const { dir } = freshConfig({ providers: { relay: { + adapter: "openai-chat", baseUrl: "https://example.com/v1", + models: ["claude-sonnet-4-6", "gpt-4o", "unknown-model"], + modelCosts: { "gpt-4o": { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } }, + } } }); + try { + const before = readFileSync(join(dir, "config.json"), "utf8"); + const result = runCli(["models", "--json"], { OPENCODEX_HOME: dir }); + expect(result.status).toBe(0); + const rows = JSON.parse(result.stdout).models; + expect(rows[0].price.cost4).toEqual({ input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }); + expect(rows[1].price).toMatchObject({ source: "user", cost4: { input: 0, output: 0 } }); + expect(rows[2].price).toBeNull(); + expect(readFileSync(join(dir, "config.json"), "utf8")).toBe(before); + const human = runCli(["models"], { OPENCODEX_HOME: dir }); + expect(human.stdout).toContain("~$3/$15 input/output per 1M tokens"); + expect(human.stdout).toContain("price unknown"); + } finally { + removeTreeWithRetry(dir); + } + }); + test("models lists all provider models", () => { const { dir } = freshConfig(); try { From e89de8f5329fe43e53e88cd0e6200e91d4a4803c Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 14:21:52 +0900 Subject: [PATCH 07/75] fix(plugins): do not trust ACL display names as root (#6019) Carried from #6019 into merge train round 3. Co-authored-by: Ingwannu --- .../src/content/docs/guides/local-plugins.md | 7 +++--- src/plugins/loader.ts | 8 ++++--- structure/ops/plugins.md | 8 ++++--- tests/lib/plugin-loader.test.ts | 24 ++++++++++++++++++- 4 files changed, 37 insertions(+), 10 deletions(-) diff --git a/docs-site/src/content/docs/guides/local-plugins.md b/docs-site/src/content/docs/guides/local-plugins.md index b952aa91419..fee1e551152 100644 --- a/docs-site/src/content/docs/guides/local-plugins.md +++ b/docs-site/src/content/docs/guides/local-plugins.md @@ -31,9 +31,10 @@ Put plugin files in `plugins/` inside the opencodex home (`~/.opencodex/plugins/ `chmod go-w ~/.opencodex/plugins ~/.opencodex/plugins/*`; on systems whose default umask is `002`, check the parent directories too. On macOS, an ACL grant to another user or group that can write, delete, change permissions, or add/remove path entries blocks loading, even if the - mode is `0600`; inspect with `ls -le`. Read-only, deny, inheritance-only, and grants only to - the path owner, the running user, or root do not block loading. On Linux, extended ACLs are - checked when `getfacl` is installed. Without it, only owner and mode bits are verified. + mode is `0600`; inspect the path itself with `/bin/ls -lebd -- `. Read-only, deny, + inheritance-only, and grants only to the path owner or running user do not block loading. ACL + display names such as `root` or `0` are not treated as numeric UID proof. On Linux, extended + ACLs are checked when `getfacl` is installed. Without it, only owner and mode bits are verified. - On Windows automatic plugin loading is disabled until an ACL trust check is available. Restart the proxy after adding, changing or removing a plugin (`ocx service restart`, or stop and diff --git a/src/plugins/loader.ts b/src/plugins/loader.ts index b1b3e37adac..a9c2ffc57c7 100644 --- a/src/plugins/loader.ts +++ b/src/plugins/loader.ts @@ -81,10 +81,12 @@ export function macAclListingTrustError(listing: string, currentUser = userInfo( if (!entry) { unparseable = true; continue; } if (entry[2] === "deny") continue; const principal = entry[1]!; - // The file owner, this process's user, and root already control the path without an ACE. - // Require the `user:` prefix: a bare or group principal might include other users. + // The file owner and this process's user already control the path without an ACE. Require the + // `user:` prefix: a bare or group principal might include other users. macOS `ls` prints a + // resolved directory-record NAME here, so neither `0` nor `root` proves that the ACE is UID 0. + // On a genuinely root-owned path, `root` is still admitted by the owner comparison. if (principal.startsWith("user:") - && [owner, currentUser, "root", "0"].includes(principal.slice(5))) continue; + && [owner, currentUser].includes(principal.slice(5))) continue; const rights = entry[3]!.split(","); if (rights.includes("only_inherit")) continue; if (rights.some(right => !MAC_ACL_BENIGN_TOKENS.has(right))) return "has an access control list"; diff --git a/structure/ops/plugins.md b/structure/ops/plugins.md index 45886a169fb..263db4c938b 100644 --- a/structure/ops/plugins.md +++ b/structure/ops/plugins.md @@ -20,9 +20,11 @@ or signs them. can swap a checked path before it is imported; files are imported through the resolved directory. On macOS, `ls -lebd` must show no effective non-owner ACL grant that can write, delete, change permissions, or add/remove path entries on the file, plugin directory, or any ancestor. Denials, - grants only to the path owner, the running user, or root, read-only grants, and inheritance-only - entries on the inspected path are safe; inherited grants effective on a descendant are checked - at that descendant. A timed-out macOS inspection retries once only if its output is empty: + grants only to the path owner or running user, read-only grants, and inheritance-only entries on + the inspected path are safe; inherited grants effective on a descendant are checked at that + descendant. `ls` renders UUID-backed principals as Directory Services record names, so names + such as `root` or `0` never establish UID 0; root-owned paths still pass through the owner check. + A timed-out macOS inspection retries once only if its output is empty: observed unsafe grants refuse immediately, and any other partial output is incomplete and also refuses loading. Unknown grants or other inspection errors also refuse loading. Linux uses `getfacl` when installed and refuses extended diff --git a/tests/lib/plugin-loader.test.ts b/tests/lib/plugin-loader.test.ts index 1ec5dda1fde..48078a3a8b3 100644 --- a/tests/lib/plugin-loader.test.ts +++ b/tests/lib/plugin-loader.test.ts @@ -137,9 +137,31 @@ test("recorded macOS ls output rejects effective non-owner write grants", () => + " 0: group:everyone allow write\n"; const ownerNamedGroup = "drwx------@ 2 runner staff 64 Sep 27 07:50 /private/var/folders/ab/tmp/plugins\n" + " 0: group:runner allow add_file\n"; - for (const listing of [pluginDir, ownedAncestor, inheritedChild, pluginFile, ownerNamedGroup]) { + // `/bin/ls -lebd` renders resolved ACL record names, not numeric UIDs. A foreign record named + // `0` must not inherit root trust merely because its name looks like UID 0 (#6017). + const numericRecordName = "-rw-------@ 1 runner staff 64 Sep 27 07:50 /plugins/plugin.ts\n" + + " 0: user:0 allow write\n"; + const rootRecordName = "-rw-------@ 1 runner staff 64 Sep 27 07:50 /plugins/plugin.ts\n" + + " 0: user:root allow write\n"; + const unresolvedUuid = "-rw-------@ 1 runner staff 64 Sep 27 07:50 /plugins/plugin.ts\n" + + " 0: user:8D95C9F2-3B29-4B30-8932-C43D3AABC123 allow write\n"; + for (const listing of [ + pluginDir, ownedAncestor, inheritedChild, pluginFile, ownerNamedGroup, + numericRecordName, rootRecordName, unresolvedUuid, + ]) { expect(macAclListingTrustError(listing)).toBe("has an access control list"); } + // Bare principals are not identity-bearing user records. Keep the rights policy explicit so a + // future parser cleanup cannot accidentally grant them the owner/current-user exemption. + const bareBenignPrincipal = "-rw-------@ 1 runner staff 64 Sep 27 07:50 /plugins/plugin.ts\n" + + " 0: runner allow read\n"; + expect(macAclListingTrustError(bareBenignPrincipal)).toBeNull(); + const bareWritePrincipal = "-rw-------@ 1 runner staff 64 Sep 27 07:50 /plugins/plugin.ts\n" + + " 0: runner allow write\n"; + expect(macAclListingTrustError(bareWritePrincipal)).toBe("has an access control list"); + const numericCurrentUser = "-rw-------@ 1 0 staff 64 Sep 27 07:50 /plugins/plugin.ts\n" + + " 0: user:0 allow write\n"; + expect(macAclListingTrustError(numericCurrentUser, "0")).toBeNull(); expect(macAclListingTrustError("drwxr-xr-x+ 23 root wheel 736 Sep 27 07:50 /\n 0: unrecognized ACL entry\n")) .toBe("access control list inspection failed"); expect(macAclListingTrustError(`${pluginDir.split("\n")[0]}\n 0: group:everyone allow future_permission\n`)) From 14913d98ad16752933e5817a9d2022021289fd77 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:22:01 +0900 Subject: [PATCH 08/75] test(plugins): pin the current user in the foreign ACL principal cases --- tests/lib/plugin-loader.test.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tests/lib/plugin-loader.test.ts b/tests/lib/plugin-loader.test.ts index 48078a3a8b3..fc89aa93a75 100644 --- a/tests/lib/plugin-loader.test.ts +++ b/tests/lib/plugin-loader.test.ts @@ -149,7 +149,9 @@ test("recorded macOS ls output rejects effective non-owner write grants", () => pluginDir, ownedAncestor, inheritedChild, pluginFile, ownerNamedGroup, numericRecordName, rootRecordName, unresolvedUuid, ]) { - expect(macAclListingTrustError(listing)).toBe("has an access control list"); + // Pin the current user so a runner whose login is `root` or a record named `0` cannot turn + // these foreign-principal refusals into the current-user exemption. + expect(macAclListingTrustError(listing, "runner")).toBe("has an access control list"); } // Bare principals are not identity-bearing user records. Keep the rights policy explicit so a // future parser cleanup cannot accidentally grant them the owner/current-user exemption. From 91267a34bef55315749d87ea69e53f152ec86365 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 14:22:06 +0900 Subject: [PATCH 09/75] fix(desktop): preserve stopped intent after update failure (#6041) Carried from #6041 into merge train round 3. Co-authored-by: Ingwannu --- desktop/src-tauri/src/exit.rs | 105 ++++++++++++++++-- desktop/src-tauri/src/updater.rs | 30 ++++- .../ADR-6033-desktop-update-intent.md | 12 ++ structure/desktop-shell.md | 12 +- 4 files changed, 145 insertions(+), 14 deletions(-) create mode 100644 structure/decisions/ADR-6033-desktop-update-intent.md diff --git a/desktop/src-tauri/src/exit.rs b/desktop/src-tauri/src/exit.rs index eac40099499..5ee8e4bff10 100644 --- a/desktop/src-tauri/src/exit.rs +++ b/desktop/src-tauri/src/exit.rs @@ -146,6 +146,14 @@ pub struct Supervision { pub reason_set: bool, } +/// State restored when an update's coordinated restart is abandoned. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct AbortedRestart { + pub phase: ExitPhase, + /// Whether the person wanted a runtime before the drain or requested one while it ran. + pub runtime_was_wanted: bool, +} + impl Supervision { /// Nothing is in flight, nobody asked for the runtime to stop, and the app is not ending. pub fn allowed(self) -> bool { @@ -161,6 +169,14 @@ struct Inner { deferred: bool, /// See [`Supervision::wanted`]. Sticky: finishing a stop does not restore it. wanted: bool, + /// The intent an update temporarily replaced with `wanted=false`; consumed if it aborts. + restart_wanted: Option, +} + +fn remember_restart_intent(inner: &mut Inner, reason: ExitReason, prior_wanted: bool) { + if reason == ExitReason::CoordinatedRestart { + inner.restart_wanted.get_or_insert(prior_wanted); + } } /// The exit sequence's state, managed by the app. @@ -179,6 +195,7 @@ impl ExitCoordinator { hides_to_tray: TrayAvailability::assumed().hides_to_tray(), deferred: false, wanted: true, + restart_wanted: None, }), } } @@ -215,17 +232,20 @@ impl ExitCoordinator { /// [`ExitCoordinator::finish_stop`] is holding the phase. pub fn claim_drain(&self, fallback: ExitReason) -> Option { let mut inner = self.inner(); + let prior_wanted = inner.wanted; // A quit or an update is on its way, whoever ends up running the drain: the runtime it // stops is not one to bring back. inner.wanted = false; match inner.phase { ExitPhase::Idle => { let reason = *inner.reason.get_or_insert(fallback); + remember_restart_intent(&mut inner, reason, prior_wanted); inner.phase = ExitPhase::Draining; Some(reason) } ExitPhase::Spawning | ExitPhase::Stopping => { - inner.reason.get_or_insert(fallback); + let reason = *inner.reason.get_or_insert(fallback); + remember_restart_intent(&mut inner, reason, prior_wanted); inner.deferred = true; None } @@ -235,6 +255,7 @@ impl ExitCoordinator { // is the one thing a user with a runtime that would not stop cannot easily do. ExitPhase::DrainFailed | ExitPhase::OwnershipUnknown => { let reason = *inner.reason.get_or_insert(fallback); + remember_restart_intent(&mut inner, reason, prior_wanted); inner.phase = ExitPhase::Draining; Some(reason) } @@ -256,10 +277,12 @@ impl ExitCoordinator { /// An update drains before it installs. When the install then fails, or the drain itself did, /// the drain's phase used to be the end of the road: `Drained` is terminal, so no runtime could /// be started again and a bare window close quit the app. This returns the app to `Idle` with - /// no claimed reason and wants a runtime again, which is what a successful update would have - /// ended in too. It touches nothing unless an update's restart holds the phase: a quit is never - /// aborted, and a drain still running belongs to whoever runs it. Returns the phase it left. - pub fn abort_restart(&self) -> Option { + /// no claimed reason and restores the runtime intent the update temporarily suppressed. A + /// retry requested while the drain was in flight wins too: aborting an older update must not + /// overwrite newer user intent. It touches nothing unless an update's restart holds the phase: + /// a quit is never aborted, and a + /// drain still running belongs to whoever runs it. Returns the phase and restored intent. + pub fn abort_restart(&self) -> Option { let mut inner = self.inner(); let left = inner.phase; let restart = inner.reason == Some(ExitReason::CoordinatedRestart); @@ -273,8 +296,15 @@ impl ExitCoordinator { inner.phase = ExitPhase::Idle; inner.reason = None; inner.deferred = false; - inner.wanted = true; - Some(left) + // `resume` can arrive after the update captured its original intent. Preserve that newer + // request as well as the older snapshot; otherwise the abort races the startup retry and + // can leave a runtime stopped even though the person just asked for it. + let runtime_was_wanted = inner.wanted || inner.restart_wanted.take().unwrap_or(false); + inner.wanted = runtime_was_wanted; + Some(AbortedRestart { + phase: left, + runtime_was_wanted, + }) } /// What the runtime supervisor reads before it acts. @@ -624,7 +654,7 @@ fn hide_windows(app: &AppHandle) { #[cfg(test)] mod tests { use super::{ - decide, DrainVerdict, ExitCoordinator, ExitDecision, ExitPhase, ExitReason, + decide, AbortedRestart, DrainVerdict, ExitCoordinator, ExitDecision, ExitPhase, ExitReason, RestartReadiness, Supervision, }; use crate::tray_availability::TrayAvailability; @@ -707,7 +737,13 @@ mod tests { ); coordinator.finish_drain(verdict); let left = coordinator.phase(); - assert_eq!(coordinator.abort_restart(), Some(left)); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: left, + runtime_was_wanted: true, + }) + ); assert_eq!(coordinator.phase(), ExitPhase::Idle); // A bare close hides again instead of quitting out of a terminal phase. assert_eq!(coordinator.decision(), ExitDecision::Hide); @@ -716,6 +752,57 @@ mod tests { } } + #[test] + fn a_failed_update_preserves_a_completed_tray_stop() { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + assert!(!coordinator.supervision().wanted); + + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: false, + }) + ); + assert_eq!(coordinator.phase(), ExitPhase::Idle); + assert_eq!(coordinator.decision(), ExitDecision::Hide); + assert!(!coordinator.supervision_allowed()); + } + + #[test] + fn a_startup_retry_during_an_update_drain_is_not_overwritten_by_abort() { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + assert!(!coordinator.supervision().wanted); + + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.resume(); + // The ending claim still prevents supervision until the failed update is handed back. + assert!(!coordinator.supervision_allowed()); + coordinator.finish_drain(DrainVerdict::Drained); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: true, + }) + ); + assert!(coordinator.supervision_allowed()); + } + #[test] fn a_quit_or_a_drain_in_flight_is_never_aborted() { let coordinator = ExitCoordinator::new(); diff --git a/desktop/src-tauri/src/updater.rs b/desktop/src-tauri/src/updater.rs index 9fd3595e5be..bcaf3349533 100644 --- a/desktop/src-tauri/src/updater.rs +++ b/desktop/src-tauri/src/updater.rs @@ -1,5 +1,5 @@ use crate::{ - exit::{ExitCoordinator, ExitPhase, RestartReadiness}, + exit::{AbortedRestart, ExitCoordinator, ExitPhase, RestartReadiness}, logging, tray, }; use serde::Serialize; @@ -422,8 +422,13 @@ pub async fn install(app: &AppHandle, update: Update) -> Result<(), String> { /// Hand a failed install back to a running app. True when the drain had already stopped the /// runtime, so the startup sequence has to bring one back; a drain that failed left it running. +/// Intent captured before the drain and a newer startup retry are both authoritative. fn after_install_failure(coordinator: &ExitCoordinator) -> bool { - coordinator.abort_restart() == Some(ExitPhase::Drained) + coordinator.abort_restart() + == Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: true, + }) } fn recover_after_failed_install(app: &AppHandle) { @@ -507,6 +512,7 @@ mod tests { DesktopUpdateState, InstallClaim, UiProjection, }; use crate::exit::{DrainVerdict, ExitCoordinator, ExitDecision, ExitReason}; + use crate::tray_availability::TrayAvailability; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{mpsc, Arc}; use tauri_utils::config::BundleType; @@ -533,6 +539,26 @@ mod tests { coordinator.finish_drain(DrainVerdict::Drained); assert!(!after_install_failure(&coordinator)); assert_eq!(coordinator.decision(), ExitDecision::Proceed); + + // A completed tray Stop remains the person's intent across repeated failed updates. + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + for _ in 0..2 { + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(!after_install_failure(&coordinator)); + assert!(!coordinator.supervision_allowed()); + assert_eq!(coordinator.decision(), ExitDecision::Hide); + } + + // A newer retry wins over the stopped intent that the update captured at claim time. + coordinator.claim_drain(ExitReason::CoordinatedRestart); + coordinator.resume(); + coordinator.finish_drain(DrainVerdict::Drained); + assert!(after_install_failure(&coordinator)); + assert!(coordinator.supervision_allowed()); } #[test] diff --git a/structure/decisions/ADR-6033-desktop-update-intent.md b/structure/decisions/ADR-6033-desktop-update-intent.md new file mode 100644 index 00000000000..98dcb3d9a44 --- /dev/null +++ b/structure/decisions/ADR-6033-desktop-update-intent.md @@ -0,0 +1,12 @@ +# ADR-6033 — failed desktop updates preserve runtime intent + +- Contract owner: [Desktop shell](../desktop-shell.md) + +## Decision record + +- Purpose and intent: Recover from a failed in-app update without undoing a person's earlier tray Stop. +- Existing implementation and constraints: A coordinated restart clears `wanted` before draining so the supervisor cannot race the update. Its abort path then unconditionally restored `wanted=true`, and the updater treated every `Drained` result as a request for immediate recovery. When the runtime was already tray-stopped, no child still produced `Drained`, so installer failure restarted a runtime that had been intentionally left off. +- Alternatives considered: Never recover after installer failure; infer intent from whether a child PID existed; snapshot the pre-drain `wanted` value inside the exit coordinator. +- Chosen approach: Capture pre-drain intent under the coordinator mutex when a coordinated restart first claims the sequence. When that settled restart is aborted, combine and consume the snapshot with any newer Resume request received during the drain. Immediate updater recovery requires both a `Drained` phase and this effective `wanted=true` intent. +- Why this approach: Process presence does not express user intent: an already-stopped runtime and one drained by the update are both absent. The coordinator is the existing authority for sticky Stop/Resume intent. Combining the captured value with its current value preserves newer user intent without a second race-prone read. +- Benefits, costs and impact: Failed updates still recover a runtime they stopped, while tray-stopped sessions remain idle unless the person explicitly retries startup during the drain. Download failures, successful installer restarts, quit ownership, and in-flight drains are unchanged. The snapshot is process-local and intentionally does not persist across a successful application restart. diff --git a/structure/desktop-shell.md b/structure/desktop-shell.md index 9bc2d8db553..55673192b64 100644 --- a/structure/desktop-shell.md +++ b/structure/desktop-shell.md @@ -168,8 +168,13 @@ restart asked for after `install` is never reached, and the package would be rep runtime still serving out of those files. A drain that did not complete refuses the install and leaves the update pending. Neither that refusal nor an install that fails after the drain strands the app: `ExitCoordinator::abort_restart` takes a coordinated restart's settled drain phase back to -idle with no claimed reason, so a close hides again and Quit works, and when the drain had stopped -the runtime the startup sequence brings one back in recovery mode. A quit's drain is never aborted. +idle with no claimed reason, so a close hides again and Quit works. When the drain had stopped the +runtime and it was wanted before the update **or** requested again while draining, the startup +sequence brings one back in recovery mode. A runtime already stopped from the tray stays stopped after a failed update unless the person +explicitly requests startup while that update drain is in flight; that newer request wins over the +captured stopped intent. A quit's drain is never aborted. + +> Decision record: [ADR-6033](decisions/ADR-6033-desktop-update-intent.md) The Tauri updater also publishes a bounded desktop snapshot over its identity-bound ProxyClient. A random process-session id travels in the embedded dashboard URL, and the dashboard requests GET /api/update/badge?surface=desktop&session=. A normal browser keeps the package badge. The shell posts each updater-state change and a 60-second heartbeat; if the proxy loses the snapshot or the shell stops, the desktop badge becomes unknown after 180 seconds. This display path never installs an update or replaces the signed Tauri result. The tray shows the same pending state: macOS draws a blue child NSView dot over the template status-item image; Windows/Linux swap a generated dotted PNG when a tray host exists. The Windows base glyph is unchanged. @@ -263,7 +268,8 @@ unreadable answers never count. It covers guest runtimes and an exit event that The exit coordinator's `wanted` intent keeps this from fighting the person. It is true from launch; the tray's Stop (when it takes the phase), a quit's drain and an update's drain clear it before the runtime's exit can arrive, finishing a stop does not restore it, and the failure page's retry sets it -again. A terminal +again. A coordinated update remembers the intent it temporarily clears: an aborted update restores a +previously wanted runtime, but never turns a completed tray Stop back on. A terminal `ocx stop` of the runtime this app started clears nothing, so the app starts it again after the backoff; the tray's Stop and Quit keep it stopped. The dashboard's own Stop, in the app's window or a browser, is refused with `desktop_supervised` while the app supervises the runtime From ead97d2e36383d7891a4b2c9ffda1e3be62557fd Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 14:23:23 +0900 Subject: [PATCH 10/75] fix(claude): validate translated strict output schemas (#6006) Carried from #6006 into merge train round 3. Co-authored-by: Ingwannu --- .../src/content/docs/reference/adapters.md | 8 + scripts/test-layout/layout.json | 1 + src/adapters/anthropic-output-schema.ts | 117 +++++++++- src/claude/inbound-model-options.ts | 12 +- src/claude/inbound.ts | 2 +- ...slated-output-schema-strict-eligibility.md | 13 ++ structure/providers/chat-compat.md | 17 +- .../claude-output-schema-strict.test.ts | 215 ++++++++++++++++++ tests/fixtures/test-layout-expected.json | 1 + 9 files changed, 362 insertions(+), 24 deletions(-) create mode 100644 structure/decisions/ADR-5901-translated-output-schema-strict-eligibility.md create mode 100644 tests/claude-integration/claude-output-schema-strict.test.ts diff --git a/docs-site/src/content/docs/reference/adapters.md b/docs-site/src/content/docs/reference/adapters.md index a77af2706ce..5f15be73148 100644 --- a/docs-site/src/content/docs/reference/adapters.md +++ b/docs-site/src/content/docs/reference/adapters.md @@ -256,6 +256,14 @@ MiMo model Command Code serves. local reference remains resolvable. OpenAI envelope fields such as schema `name`, envelope `description`, and `strict` are not part of the Anthropic wire format. JSON object mode without a schema has no Anthropic equivalent and is not translated. + In the reverse direction, translated Anthropic output schemas retain the caller's original + schema. `strict: true` requires an object root without a root union and complete closed objects + throughout nested properties, array items, unions and definitions: `properties` must be an object, + `additionalProperties` must be `false`, and `required` must contain exactly its property names. + Incomplete objects, unknown schema keywords, and values outside OpenAI's documented strict + subset use explicit `strict: false`; the classifier uses an allowlist rather than chasing each + unsupported constraint separately. The proxy does not invent + required fields, close an open object, or discard the caller's schema merely to obtain strict mode. - Always sends `anthropic-version: 2023-06-01`. Streams `content_block_delta` (`text_delta`, `thinking_delta`, compatible `reasoning_delta`, `input_json_delta`). The SSE decoder preserves event state across fetch chunks and accepts a terminal `message_stop` without a trailing newline. diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index aecc89fbbc0..4856cd23048 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -477,6 +477,7 @@ "claude-inbound-cache-stabilize.test.ts": "claude-integration", "claude-inbound-debug.test.ts": "claude-integration", "claude-inbound.test.ts": "claude-integration", + "claude-output-schema-strict.test.ts": "claude-integration", "claude-intercept-integration.test.ts": "server", "claude-intercept-local-ca.test.ts": "claude-integration", "claude-intercept-model-bindings.test.ts": "claude-integration", diff --git a/src/adapters/anthropic-output-schema.ts b/src/adapters/anthropic-output-schema.ts index 122e89a8869..bc1758b30e3 100644 --- a/src/adapters/anthropic-output-schema.ts +++ b/src/adapters/anthropic-output-schema.ts @@ -136,8 +136,59 @@ export function isAnthropicOutputSchema(schema: Record): boolea } } +// Anthropic accepts `uri`, but OpenAI strict Structured Outputs documents a narrower +// string-format set. Keep this separate from SUPPORTED_STRING_FORMATS: changing the Anthropic +// normalizer to satisfy the translated OpenAI route would silently remove native guidance. +const OPENAI_STRICT_STRING_FORMATS = new Set([ + "date-time", "time", "date", "duration", "email", "hostname", "ipv4", "ipv6", "uuid", +]); + +const OPENAI_STRICT_COMMON_KEYWORDS = new Set([ + "$defs", "definitions", "$ref", "type", "title", "description", "enum", "const", "anyOf", +]); +const OPENAI_STRICT_OBJECT_KEYWORDS = new Set([ + "properties", "patternProperties", "required", "additionalProperties", +]); +const OPENAI_STRICT_STRING_KEYWORDS = new Set(["pattern", "format"]); +const OPENAI_STRICT_NUMBER_KEYWORDS = new Set([ + "multipleOf", "maximum", "exclusiveMaximum", "minimum", "exclusiveMinimum", +]); +const OPENAI_STRICT_ARRAY_KEYWORDS = new Set(["items", "minItems", "maxItems"]); +const OPENAI_STRICT_TYPES = new Set(["string", "number", "boolean", "integer", "object", "array", "null"]); +// Fine-tuned models implement a narrower documented Structured Outputs subset. Falling back to +// non-strict keeps the caller's schema intact; deleting these constraints would silently widen it. +const OPENAI_FINE_TUNED_UNSUPPORTED_KEYWORDS = new Set([ + "patternProperties", "pattern", "format", + ...OPENAI_STRICT_NUMBER_KEYWORDS, + "minItems", "maxItems", +]); + +function openAiStrictSchemaTypes(value: unknown): Set | null { + if (value === undefined) return new Set(); + const values = typeof value === "string" ? [value] : value; + if (!Array.isArray(values) || values.length === 0 + || values.some(type => typeof type !== "string" || !OPENAI_STRICT_TYPES.has(type))) return null; + const types = new Set(values); + if (types.size !== values.length) return null; + // OpenAI documents unions through anyOf; the type-array shorthand is reserved for nullable fields. + if (types.size > 1 && (types.size !== 2 || !types.has("null"))) return null; + return types; +} + +function hasOnlyOpenAiStrictKeywords(node: Record, types: Set): boolean { + for (const key of Object.keys(node)) { + if (OPENAI_STRICT_COMMON_KEYWORDS.has(key)) continue; + if (types.has("object") && OPENAI_STRICT_OBJECT_KEYWORDS.has(key)) continue; + if (types.has("string") && OPENAI_STRICT_STRING_KEYWORDS.has(key)) continue; + if ((types.has("number") || types.has("integer")) && OPENAI_STRICT_NUMBER_KEYWORDS.has(key)) continue; + if (types.has("array") && OPENAI_STRICT_ARRAY_KEYWORDS.has(key)) continue; + return false; + } + return true; +} + /** - * Does every object in this schema list ALL of its properties as required? + * Can this schema preserve its object contract under OpenAI strict mode? * * OpenAI's structured-output strict mode demands exactly that, and rejects anything else with * `'required' is required to be supplied and to be an array including every key in properties`. @@ -148,26 +199,70 @@ export function isAnthropicOutputSchema(schema: Record): boolea * would silently change the contract the caller asked for, so the only honest answer is to stop * claiming strict for these schemas -- the schema is still sent and still honoured as guidance. */ -export function satisfiesOpenAiStrictSchema(value: unknown): boolean { - if (Array.isArray(value)) return value.every(satisfiesOpenAiStrictSchema); - if (!value || typeof value !== "object") return true; - const node = value as Record; - // `allOf` is not supported under strict Structured Outputs at all, wherever it appears. - if ("allOf" in node) return false; +export function satisfiesOpenAiStrictSchema(value: unknown, fineTuned = false): boolean { + // Callers pass one schema node. Boolean/null/array schemas are outside the documented + // Structured Outputs subset; schema lists such as anyOf are validated explicitly below. + if (!isRecord(value)) return false; + const node = value; + // This is a compatibility proof, so it fails closed on unknown schema keywords instead of + // maintaining an inevitably incomplete denylist. The original schema is still forwarded. + const types = openAiStrictSchemaTypes(node.type); + if (!types || !hasOnlyOpenAiStrictKeywords(node, types)) return false; + if (fineTuned && Object.keys(node).some(key => OPENAI_FINE_TUNED_UNSUPPORTED_KEYWORDS.has(key))) { + return false; + } + if (Object.hasOwn(node, "$ref") && typeof node.$ref !== "string") return false; + if (Object.hasOwn(node, "title") && typeof node.title !== "string") return false; + if (Object.hasOwn(node, "description") && typeof node.description !== "string") return false; + if (Object.hasOwn(node, "enum") && (!Array.isArray(node.enum) || node.enum.length === 0)) return false; + if (Object.hasOwn(node, "format")) { + if (typeof node.format !== "string" || !OPENAI_STRICT_STRING_FORMATS.has(node.format)) { + return false; + } + } + if (Object.hasOwn(node, "pattern") && typeof node.pattern !== "string") return false; + for (const key of OPENAI_STRICT_NUMBER_KEYWORDS) { + const constraint = node[key]; + if (Object.hasOwn(node, key) && (typeof constraint !== "number" || !Number.isFinite(constraint))) return false; + } + for (const key of ["minItems", "maxItems"]) { + const constraint = node[key]; + if (Object.hasOwn(node, key) + && (typeof constraint !== "number" || !Number.isInteger(constraint) || constraint < 0)) return false; + } const properties = node.properties; - if (isRecord(properties)) { + const objectType = types.has("object"); + if (objectType || Object.hasOwn(node, "properties")) { // An object node must list every property in `required` AND close itself to extras. The // caller's schema is forwarded verbatim -- `isAnthropicOutputSchema` normalizes a CLONE for // its own acceptance check -- so an object that never said `additionalProperties: false` // reaches the wire without it and is refused, however complete its `required` is. - if (node.additionalProperties !== false) return false; + // Anthropic's acceptance probe fills a missing/malformed property map on its clone; + // that must not certify the unchanged bare object which actually reaches the wire. + if (!isRecord(properties) || node.additionalProperties !== false) return false; // Strict mode also requires `required` to be supplied at all, even for an empty // `properties` map, so a missing array is not the same as an empty one. if (!Array.isArray(node.required)) return false; const keys = Object.keys(properties); const required: unknown[] = node.required; const requiredKeys = new Set(required); - if (keys.some(key => !requiredKeys.has(key))) return false; + if (required.length !== keys.length || keys.some(key => !requiredKeys.has(key))) return false; + } + // Walk schema positions, not arbitrary JSON values: a property named `not` or an enum/const + // value containing `properties` is data, not another schema node to certify or reject. + for (const key of ["properties", "$defs", "definitions", "patternProperties"]) { + if (!Object.hasOwn(node, key)) continue; + const entries = node[key]; + if (!isRecord(entries) + || !Object.values(entries).every(entry => satisfiesOpenAiStrictSchema(entry, fineTuned))) return false; + } + if (Object.hasOwn(node, "items") && !satisfiesOpenAiStrictSchema(node.items, fineTuned)) return false; + if (Object.hasOwn(node, "anyOf")) { + const anyOf = node.anyOf; + if (!Array.isArray(anyOf) || anyOf.length === 0 + || !anyOf.every(entry => satisfiesOpenAiStrictSchema(entry, fineTuned))) { + return false; + } } - return Object.values(node).every(satisfiesOpenAiStrictSchema); + return true; } diff --git a/src/claude/inbound-model-options.ts b/src/claude/inbound-model-options.ts index 3441d275d0d..40e32a3157b 100644 --- a/src/claude/inbound-model-options.ts +++ b/src/claude/inbound-model-options.ts @@ -96,7 +96,13 @@ export function effortFromOutputConfig(outputConfig: unknown): string | undefine return typeof effort === "string" && OUTPUT_CONFIG_EFFORTS.has(effort) ? effort : undefined; } -export function formatFromOutputConfig(outputConfig: unknown): Rec | undefined { +function isFineTunedOpenAiTarget(model: string | undefined): boolean { + if (!model) return false; + const separator = model.lastIndexOf("/"); + return model.slice(separator + 1).startsWith("ft:"); +} + +export function formatFromOutputConfig(outputConfig: unknown, resolvedModel?: string): Rec | undefined { if (!isRec(outputConfig) || !isRec(outputConfig.format)) return undefined; const format = outputConfig.format; if ( @@ -115,7 +121,9 @@ export function formatFromOutputConfig(outputConfig: unknown): Rec | undefined { name: "response", schema: format.schema, // Strict Structured Outputs also needs an object at the root; a root anyOf/oneOf is refused. - strict: format.schema.type === "object" && satisfiesOpenAiStrictSchema(format.schema), + strict: format.schema.type === "object" + && !Object.hasOwn(format.schema, "anyOf") && !Object.hasOwn(format.schema, "oneOf") + && satisfiesOpenAiStrictSchema(format.schema, isFineTunedOpenAiTarget(resolvedModel)), }; } diff --git a/src/claude/inbound.ts b/src/claude/inbound.ts index 2aba1fdc787..1b190193362 100644 --- a/src/claude/inbound.ts +++ b/src/claude/inbound.ts @@ -460,7 +460,7 @@ function translateAnthropicRequest( if (Array.isArray(raw.stop_sequences) && raw.stop_sequences.length > 0) { body.stop = raw.stop_sequences.filter((s): s is string => typeof s === "string"); } - const outputConfigFormat = formatFromOutputConfig(raw.output_config); + const outputConfigFormat = formatFromOutputConfig(raw.output_config, body.model as string); if (outputConfigFormat) body.text = { format: outputConfigFormat }; let cacheKeySource: ClaudeCacheKeySource = null; if (isRec(raw.metadata) && typeof raw.metadata.user_id === "string") { diff --git a/structure/decisions/ADR-5901-translated-output-schema-strict-eligibility.md b/structure/decisions/ADR-5901-translated-output-schema-strict-eligibility.md new file mode 100644 index 00000000000..937becbfdc7 --- /dev/null +++ b/structure/decisions/ADR-5901-translated-output-schema-strict-eligibility.md @@ -0,0 +1,13 @@ +# ADR-5901 — decision recorded under "Anthropic structured-output compatibility" + +- Contract owner: [providers/chat-compat.md](../providers/chat-compat.md#anthropic-structured-output-compatibility) + +## Decision record + +- Purpose and intent: Avoid asserting OpenAI strict-mode compatibility for an unchanged Anthropic output schema whose objects do not satisfy the strict object contract. +- Existing implementation and constraints: The Anthropic acceptance probe normalizes a clone, including missing property maps and object closure. Translation forwards the original schema. Checking only existing property maps therefore certified a bare object that the destination would reject. Generic JSON recursion also confused property names and literal example/default objects with schema nodes. +- Alternatives considered: Normalize the forwarded schema by inventing closure and required fields; reject all such Anthropic input; retain the original schema and classify strict eligibility separately. +- Chosen approach: Preserve the schema unchanged and emit explicit non-strict mode when an object lacks a valid property map, closure or an exact required-name set, or any schema node uses a keyword/value outside OpenAI's documented subset. Use a per-type allowlist so new unsupported keywords fail closed without growing a denylist. Validate nested schema positions and nullable object types without traversing literal enum/const data. A root union remains non-strict even when accompanied by an object type. After Claude alias/model-map resolution, apply the documented narrower fine-tuned subset recursively to direct or provider-qualified `ft:` targets. Do not claim that this ingress decision predicts later provider aliases or combo selection. Keep Anthropic's broader native format acceptance separate. +- Why this approach: The caller's optional fields and open-object contract must not silently narrow just to satisfy another provider. Non-strict fallback is the existing translated-output contract; outright rejection would unnecessarily remove accepted requests. +- Benefits, costs and impact: Valid nested and empty closed objects plus documented string formats keep strict mode; incomplete or unsupported schemas no longer claim strict compatibility. Fine-tuned targets retain their schema but use non-strict mode for pattern/format, numeric, array-size and pattern-property constraints; otherwise eligible fine-tuned schemas remain strict. The helper is a compatibility classifier, not a complete JSON Schema validator or proof of every eventual routed model. Native Anthropic normalization and unrelated non-object/non-strict handling remain unchanged. +- Contract reference: [OpenAI Structured Outputs supported schemas](https://developers.openai.com/api/docs/guides/structured-outputs#supported-schemas), checked 2026-09-26. The documented unsupported composition keywords are `allOf`, `not`, `dependentRequired`, `dependentSchemas`, `if`, `then` and `else`; only `anyOf`, not `oneOf`, is in the supported type list. Other schema positions outside that documented subset also fall back to non-strict mode. diff --git a/structure/providers/chat-compat.md b/structure/providers/chat-compat.md index 9aaeceababe..b1eea210ff9 100644 --- a/structure/providers/chat-compat.md +++ b/structure/providers/chat-compat.md @@ -439,18 +439,15 @@ family shared by unrelated upstreams. ## Anthropic structured-output compatibility -The Anthropic adapter lowers Responses `text.format` and Chat Completions `response_format` JSON -Schema requests to `output_config.format`. The local transform follows Anthropic's TypeScript SDK -subset so upstream rejects neither OpenAI-only envelope fields nor unsupported schema constraints. -The adapter merges `format` into an existing adaptive-thinking `output_config` rather than replacing -it, so a compatible `output_config.effort` remains alongside the structured-output format. -Routed Anthropic Messages input carries `output_config.format` through internal `text.format`, so -stored-OAuth requests regain the same native format when the Anthropic adapter rebuilds the wire body. -Unsupported constraints remain in `description` as model guidance instead of disappearing. Root -`$defs` stay beside a root `$ref`, intentionally differing from the current SDK transform's early -`$ref` return so local references remain resolvable. +The Anthropic adapter lowers Responses `text.format` and Chat Completions `response_format` JSON Schema requests to `output_config.format`, following Anthropic's TypeScript SDK subset. +It merges `format` into the existing adaptive-thinking `output_config`, preserving compatible `output_config.effort`. Unsupported constraints remain in `description` as model guidance; root `$defs` remain beside a root `$ref` so local references resolve. +Routed Anthropic Messages input carries `output_config.format` through internal `text.format`; stored-OAuth requests regain that format when the Anthropic adapter rebuilds the wire body. +In that inbound direction, `src/adapters/anthropic-output-schema.ts` checks the original schema, not the normalized acceptance clone. Strict mode requires an object root without a root union, and every object must supply a property map, `additionalProperties: false`, and exactly matching `required` names. +The check traverses schema-valued properties, array items, unions and definitions; property names and literal enum/const values are not schema nodes. It admits only documented schema keywords for the node's declared type, so incomplete objects, unknown constraints and invalid keyword values retain their original schema with explicit `strict: false` instead of waiting for a denylist update. Fine-tuned `ft:` targets use OpenAI's narrower model-specific keyword subset after Claude alias/model-map resolution; later provider/combo routing remains outside this ingress proof. Anthropic's native acceptance still includes `uri`; only the translated OpenAI strict claim uses the narrower set. +`tests/claude-integration/claude-output-schema-strict.test.ts` covers the acceptance probe, inbound format, translated Responses parser, recursive object controls and caller-schema preservation. > Decision record: [ADR-0066](../decisions/ADR-0066-anthropic-structured-output-compatibility.md) +> Decision record: [ADR-5901](../decisions/ADR-5901-translated-output-schema-strict-eligibility.md) ## Reasoning display parity (hideThinkingSummary) diff --git a/tests/claude-integration/claude-output-schema-strict.test.ts b/tests/claude-integration/claude-output-schema-strict.test.ts new file mode 100644 index 00000000000..f5ffc6dc635 --- /dev/null +++ b/tests/claude-integration/claude-output-schema-strict.test.ts @@ -0,0 +1,215 @@ +import { describe, expect, test } from "bun:test"; +import { isAnthropicOutputSchema, satisfiesOpenAiStrictSchema } from "../../src/adapters/anthropic-output-schema"; +import { anthropicToResponsesBody } from "../../src/claude/inbound"; +import { formatFromOutputConfig } from "../../src/claude/inbound-model-options"; +import { parseRequest } from "../../src/responses/parser"; + +type Schema = Record; + +function closedObject(properties: Schema = {}): Schema { + return { type: "object", properties, required: Object.keys(properties), additionalProperties: false }; +} + +function expectTranslatedStrict(schema: Schema, strict: boolean): void { + const before = structuredClone(schema); + // Acceptance normalizes a clone; strict eligibility must describe the ORIGINAL wire schema. + expect(isAnthropicOutputSchema(schema)).toBe(true); + const format = formatFromOutputConfig({ format: { type: "json_schema", schema } }); + expect(format).toEqual({ type: "json_schema", name: "response", schema, strict }); + expect(format?.schema).toBe(schema); + const body = anthropicToResponsesBody({ + model: "claude-sonnet-5", max_tokens: 256, + messages: [{ role: "user", content: "Return JSON" }], + output_config: { format: { type: "json_schema", schema } }, + }); + expect(parseRequest(body).options.textFormat).toEqual({ type: "json_schema", name: "response", schema, strict }); + expect(schema).toEqual(before); +} + +describe("translated Anthropic structured output strict eligibility (#5901 follow-up)", () => { + test.each([ + { type: "object" }, + { type: "object", additionalProperties: false, required: [] }, + { type: "object", properties: null, additionalProperties: false, required: [] }, + { type: "object", properties: [], additionalProperties: false, required: [] }, + { type: "object", properties: "invalid", additionalProperties: false, required: [] }, + { type: "object", properties: {} }, + { type: "object", properties: {}, required: [] }, + { type: "object", properties: {}, required: [], additionalProperties: true }, + { type: "object", properties: {}, additionalProperties: false }, + ])("does not certify an incomplete object: %j", schema => { + expect(satisfiesOpenAiStrictSchema(schema)).toBe(false); + expectTranslatedStrict(schema, false); + }); + + test.each([undefined, null, "answer", [], ["answer", "extra"], ["answer", "answer"], [123]] + .map(required => ({ required })))( + "requires exactly the declared property names: %j", ({ required }) => { + const schema = { ...closedObject({ answer: { type: "string" } }), required }; + expect(satisfiesOpenAiStrictSchema(schema)).toBe(false); + expectTranslatedStrict(schema, false); + }, + ); + + test.each([ + { payload: null }, + { payload: { type: "object" } }, + { payload: { type: ["object", "null"] } }, + { payload: { type: "array", items: { type: "object" } } }, + { payload: { anyOf: [{ type: "null" }, { type: "object" }] } }, + { payload: { anyOf: [null, { type: "string" }] } }, + ])("checks object contracts in nested schema positions: %j", properties => { + const schema = closedObject(properties); + expect(satisfiesOpenAiStrictSchema(schema)).toBe(false); + expectTranslatedStrict(schema, false); + }); + + test("checks object contracts inside referenced definitions", () => { + const schema = { + ...closedObject({ payload: { $ref: "#/$defs/payload" } }), + $defs: { payload: { type: "object" } }, + }; + expectTranslatedStrict(schema, false); + }); + + test.each(["allOf", "oneOf", "not", "dependentRequired", "dependentSchemas", "if", "then", "else", + "additionalItems", "contains", "minContains", "maxContains", "prefixItems", "uniqueItems", + "propertyNames", "minProperties", "maxProperties", "unevaluatedItems", "unevaluatedProperties"])( + "does not claim unsupported strict keyword %s at a schema node", keyword => { + const unsupported = { + ...closedObject(), + [keyword]: keyword === "allOf" ? [closedObject()] : keyword === "uniqueItems" ? true : {}, + }; + expectTranslatedStrict(unsupported, false); + expectTranslatedStrict(closedObject({ payload: { type: "array", items: unsupported } }), false); + }, + ); + + test("an unsupported array constraint on a property survives with strict disabled", () => { + const schema = closedObject({ + tags: { type: "array", items: { type: "string" }, uniqueItems: true }, + }); + expectTranslatedStrict(schema, false); + }); + + test("unknown schema keywords fail closed instead of waiting for another denylist entry", () => { + for (const property of [ + { type: "string", contentEncoding: "base64" }, + { type: "string", minLength: 1 }, + { type: "string", default: "value" }, + { type: "string", examples: ["value"] }, + { type: "string", readOnly: true }, + ]) { + expectTranslatedStrict(closedObject({ value: property }), false); + } + }); + + test("OpenAI strict string formats use the documented subset without changing Anthropic acceptance", () => { + const uriSchema = closedObject({ resource: { type: "string", format: "uri" } }); + expectTranslatedStrict(uriSchema, false); + expect((uriSchema.properties as Schema).resource).toEqual({ type: "string", format: "uri" }); + + for (const format of [ + "date-time", "time", "date", "duration", "email", "hostname", "ipv4", "ipv6", "uuid", + ]) { + expectTranslatedStrict(closedObject({ value: { type: "string", format } }), true); + } + }); + + test("keeps complete empty, nested, array and nullable objects strict", () => { + expectTranslatedStrict(closedObject(), true); + const schema = { + ...closedObject({ + empty: closedObject(), + rows: { type: "array", items: closedObject({ answer: { type: "string" } }) }, + optionalValue: { ...closedObject(), type: ["object", "null"] }, + choice: { anyOf: [closedObject(), { type: "null" }] }, + referenced: { $ref: "#/$defs/payload" }, + }), + $defs: { payload: closedObject({ next: { $ref: "#/$defs/payload" } }) }, + }; + expectTranslatedStrict(schema, true); + }); + + test("keeps documented type-specific constraints strict", () => { + expectTranslatedStrict(closedObject({ + text: { type: "string", pattern: "^[a-z]+$", format: "hostname" }, + count: { type: "integer", minimum: 0, exclusiveMaximum: 10, multipleOf: 2 }, + rows: { type: "array", items: { type: "boolean" }, minItems: 1, maxItems: 3 }, + }), true); + }); + + test("fine-tuned targets preserve unsupported constraints with strict disabled", () => { + const schema = closedObject({ + nested: { + anyOf: [ + { type: "string", pattern: "^[a-z]+$" }, + { type: "array", items: { type: "integer", minimum: 0 }, maxItems: 3 }, + ], + }, + mapped: { $ref: "#/$defs/mapped" }, + }); + schema.$defs = { + mapped: { + type: "object", properties: {}, required: [], additionalProperties: false, + patternProperties: { "^x-": { type: "string" } }, + }, + }; + const before = structuredClone(schema); + + for (const model of ["ft:gpt-4.1-nano:org::name", "openai/ft:gpt-4.1-nano:org::name"]) { + expect(formatFromOutputConfig({ format: { type: "json_schema", schema } }, model)) + .toEqual({ type: "json_schema", name: "response", schema, strict: false }); + } + expect(formatFromOutputConfig({ format: { type: "json_schema", schema } }, "gpt-4.1-nano")) + .toEqual({ type: "json_schema", name: "response", schema, strict: true }); + expect(schema).toEqual(before); + }); + + test("resolved modelMap targets control fine-tuned strict eligibility", () => { + const constrained = closedObject({ value: { type: "string", format: "email" } }); + const raw = { + model: "claude-sonnet-5", max_tokens: 256, + messages: [{ role: "user", content: "Return JSON" }], + output_config: { format: { type: "json_schema", schema: constrained } }, + }; + const fineTuned = anthropicToResponsesBody(raw, { + modelMap: { "claude-sonnet-5": "openai/ft:gpt-4.1-nano:org::name" }, + }); + expect((fineTuned.text as { format: { strict: boolean } }).format.strict).toBe(false); + expect(fineTuned.model).toBe("openai/ft:gpt-4.1-nano:org::name"); + + const ordinary = anthropicToResponsesBody({ ...raw, output_config: { + format: { type: "json_schema", schema: closedObject({ value: { type: "string" } }) }, + } }, { + modelMap: { "claude-sonnet-5": "openai/ft:gpt-4.1-nano:org::name" }, + }); + expect((ordinary.text as { format: { strict: boolean } }).format.strict).toBe(true); + }); + + test("property names and literal data are not mistaken for schema keywords", () => { + const schema = closedObject({ + not: { type: "string" }, allOf: { type: "string" }, properties: { type: "string" }, + example: { + ...closedObject({ type: { type: "string" } }), + enum: [{ type: "object", not: {}, properties: null }], + }, + }); + expectTranslatedStrict(schema, true); + }); + + test.each(["anyOf", "oneOf"])("a root %s stays non-strict even with an explicit object type", keyword => { + expectTranslatedStrict({ ...closedObject(), [keyword]: [closedObject()] }, false); + }); + + test.each([{ type: "string" }, { type: "array", items: { type: "string" } }])( + "keeps non-object root behavior: %j", schema => expectTranslatedStrict(schema, false), + ); + + test("preserves optional fields and unsupported-schema rejection", () => { + expectTranslatedStrict({ ...closedObject({ answer: { type: "string" } }), required: [] }, false); + expect(formatFromOutputConfig({ format: { type: "json_schema", schema: { description: "no type" } } })) + .toBeUndefined(); + expect(formatFromOutputConfig({ format: { type: "json_object" } })).toBeUndefined(); + }); +}); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index b44e92dbfec..d8bfa12451f 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -313,6 +313,7 @@ "claude-inbound-cache-stabilize.test.ts": "claude-integration", "claude-inbound-debug.test.ts": "claude-integration", "claude-inbound.test.ts": "claude-integration", + "claude-output-schema-strict.test.ts": "claude-integration", "claude-intercept-integration.test.ts": "server", "claude-intercept-local-ca.test.ts": "claude-integration", "claude-intercept-model-bindings.test.ts": "claude-integration", From 8f25206f3edbe73ae050b0677813d4d36709ea17 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 14:23:34 +0900 Subject: [PATCH 11/75] test(responses): pin established websocket fallback (#6011) Carried from #6011 into merge train round 3. Co-authored-by: Ingwannu --- .../docs/reference/configuration/server.md | 5 + ...ADR-4191-established-websocket-fallback.md | 14 +++ structure/transports/responses-failover.md | 11 ++ tests/responses/ws-ambiguous-resend.test.ts | 102 +++++++++++++++++- 4 files changed, 130 insertions(+), 2 deletions(-) create mode 100644 structure/decisions/ADR-4191-established-websocket-fallback.md diff --git a/docs-site/src/content/docs/reference/configuration/server.md b/docs-site/src/content/docs/reference/configuration/server.md index a9b694c93a3..c8e9f967ffb 100644 --- a/docs-site/src/content/docs/reference/configuration/server.md +++ b/docs-site/src/content/docs/reference/configuration/server.md @@ -90,6 +90,11 @@ refusal returns as soon as that grant is spent, the leg has no send left, or a r for any other reason. A request that already emitted output or a tool call keeps the refusal regardless. A caller that cancels mid-replacement gets the cancellation, not the refusal. +For WebSocket recovery, “before the first Responses event” is stricter than “before the +first text”: even `response.created`, a tool event or a usage-bearing response closes the +replacement window. Ping/pong and quota metadata alone do not. A later socket close cannot +turn an already settled cancellation or connect/silence timeout into an HTTP retry. + `noProxy` accepts either a comma-separated string or an array. Both forms add entries without replacing an inherited `NO_PROXY`: diff --git a/structure/decisions/ADR-4191-established-websocket-fallback.md b/structure/decisions/ADR-4191-established-websocket-fallback.md new file mode 100644 index 00000000000..1dec5e87b9f --- /dev/null +++ b/structure/decisions/ADR-4191-established-websocket-fallback.md @@ -0,0 +1,14 @@ +# ADR-4191 — decision recorded under "Ambiguous-resend gate" + +- Contract owner: [transports/responses-failover.md](../transports/responses-failover.md#ambiguous-resend-gate) +- Recorded: 2026-09-26 + +## Decision record + +- Purpose and intent: Recover an established Codex WebSocket that closes or errors before any semantic Responses event without replaying partially observed turns. +- Existing implementation and constraints: At dev `5518653a9a`, the implementation from `aed3bb8f42` already marks eligible socket deaths and asks the shared ambiguous-resend gate in passthrough dispatch. A successful WebSocket send can have started inference even if nothing came back. Liveness/prelude diagnostics and silence deadlines already exist independently. +- Alternatives considered: Add an unconditional SSE retry inside the exchange; add a second transport-local retry counter; replace requests after text-free response events; or retain the existing dispatch-owned replacement and strengthen regression coverage. +- Chosen approach: Keep the existing dispatch-owned HTTP-only replacement. Require the provider's `retryOnReset` policy, self-contained request judgment, unspent request-wide grant and available send budget. Record and charge the physical send at the ordinary boundary, after credential selection is revalidated. Extend deterministic coverage rather than add a duplicate transport path. +- Why: The exchange cannot independently authorize a new inference, reserve spend, refresh credentials or choose another account. The shared dispatch already owns those decisions and the request identity. Any semantic response event ends eligibility, including creation, tool and usage events before text. Quota control frames and pong events indicate liveness only. +- Advantages, costs and consequences: One default replacement remains available without weakening hosted-tool, cancellation, timeout, native-control or committed-output exclusions. The opt-in still accepts possible duplicate inference billing; absence of output does not prove non-execution. An explicit `replacements: 2` retains the existing shared-policy contract for a subsequent HTTP reset, rather than becoming a second WebSocket fallback. This change does not widen that policy or alter runtime behavior. +- Validation boundary: Focused tests are authored but not executed in this work item because the operator prohibited test/build/install execution. Existing implementation and source traces are evidence of placement, not claims that the added assertions pass. diff --git a/structure/transports/responses-failover.md b/structure/transports/responses-failover.md index 67e512b608e..0d5d4a310fd 100644 --- a/structure/transports/responses-failover.md +++ b/structure/transports/responses-failover.md @@ -105,6 +105,17 @@ is sorted exactly like the pre-header row's (see [ambiguous connection-reset replay boundary](#ambiguous-connection-reset-replay-boundary)) before it goes round the recovery loop again. +The boundary is the first non-control Responses event, not the first visible text delta: +`response.created`, output/tool events and response usage all close the WebSocket replacement +window. Quota metadata and ping/pong liveness alone do not. Cancellation and connect/silence +deadlines never acquire the socket-death marker, even if a late close follows them. +`tests/responses/ws-ambiguous-resend.test.ts` covers these boundaries through the exchange and +the existing HTTP-only dispatch, including request-field preservation and terminal fallback +answers. The replacement uses the shared credential-selection guard and physical-send ledger; +there is no transport-local retry budget or credential snapshot with independent authority. + +> Decision record: [ADR-4191](../decisions/ADR-4191-established-websocket-fallback.md) + ## Console upload rejection recovery `src/providers/opencode-zen-rate-limit.ts` recognizes the complete Console upload-rejection envelope only at the effective HTTPS opencode.ai Zen/Go generation endpoint. A provider row name cannot authorize another destination. The two recovery loops in `src/server/responses/core.ts` wait 800 ms and replay the captured serialized request once; cancellation, nonreplayable responses, other errors and a second upload rejection keep their failure semantics. The recovery kind is persisted as `console-go-upload-retry` and has a localized Logs label. diff --git a/tests/responses/ws-ambiguous-resend.test.ts b/tests/responses/ws-ambiguous-resend.test.ts index 38ddd5b8edb..36e12ae4cb1 100644 --- a/tests/responses/ws-ambiguous-resend.test.ts +++ b/tests/responses/ws-ambiguous-resend.test.ts @@ -116,7 +116,7 @@ const QUOTA_FRAME = JSON.stringify({ type: "codex.rate_limits", rate_limits: { primary: { used_percent: 10, window_minutes: 10080 } }, }); -/** The three ways a socket can die under the send before anything was promised to the client. */ +/** Control traffic proves liveness, not inference output or safe non-delivery. */ const SOCKET_DEATHS: Array<[string, (ws: FakeWebSocket) => void, "pre-header" | "protocol-prelude"]> = [ ["nothing came back", ws => { ws.emit("open", {}); @@ -131,6 +131,13 @@ const SOCKET_DEATHS: Array<[string, (ws: FakeWebSocket) => void, "pre-header" | ws.emit("open", {}); ws.emit("error", {}); }, "pre-header"], + ["pongs and metadata arrived without a Responses event", ws => { + ws.emit("open", {}); + ws.emit("pong", {}); + ws.emit("message", { data: QUOTA_FRAME }); + ws.emit("message", { data: JSON.stringify({ type: "codex.response.metadata", headers: {} }) }); + ws.emit("close", { code: 1006 }); + }, "protocol-prelude"], ]; const noFallback = (async () => { @@ -176,6 +183,23 @@ describe("the exchange records a socket that died under the send (#4191)", () => await expect(response.text()).rejects.toThrow("closed before a Responses terminal event"); }); + test("a connect deadline after send cannot become a socket-death replacement", async () => { + const abort = new AbortController(); + installFake(ws => { + ws.emit("open", {}); + abort.abort(new DOMException("connect deadline", "TimeoutError")); + // A late close must not replace the already settled deadline verdict. + ws.emit("close", { code: 1006 }); + }); + const response = await codexWsUpstreamFetch( + CODEX_URL, { ...streamingInit(), signal: abort.signal }, noFallback, + ); + expect(response.status).toBe(504); + expect(codexWsSocketDeathStage(response)).toBeUndefined(); + expect(readCodexWsStage(response)?.sent).toBe(true); + expect(FakeWebSocket.instances[0]!.sent).toHaveLength(1); + }); + test("a steering exchange's death is not offered: its channel may have sent more than the create", async () => { installFake(ws => { ws.emit("open", {}); @@ -251,9 +275,10 @@ describe("handleResponses replaces a dead socket's send once under retryOnReset config: OcxConfig, logCtx: RequestLogContext = { model: "", provider: "" }, sendBudget = createRequestExecutionBudget(), + abortSignal?: AbortSignal, ): Promise { takeSpendHome(); - return handleResponses(request, config, logCtx, { codexWsRuntimeIdentity: BOUNDED_WS_RUNTIME, sendBudget }); + return handleResponses(request, config, logCtx, { codexWsRuntimeIdentity: BOUNDED_WS_RUNTIME, sendBudget, abortSignal }); } describe("in pool mode", () => { @@ -382,6 +407,7 @@ describe("handleResponses replaces a dead socket's send once under retryOnReset test.each([ ["the provider grants nothing", {}, {}], ["the turn is stored upstream", { retryOnReset: {} }, { store: true }], + ["the request declares an upstream hosted tool", { retryOnReset: {} }, { tools: [{ type: "web_search" }] }], ])("the 502 stands and nothing else is sent when %s", async (_name, provider, body) => { installFake(SOCKET_DEATHS[0]![1]); const http = stubHttp(completed); @@ -392,6 +418,78 @@ describe("handleResponses replaces a dead socket's send once under retryOnReset expect(http).toHaveLength(0); }); + test.each([ + { type: "response.created", response: { id: "r-ws", status: "in_progress" } }, + { type: "response.output_text.delta", response_id: "r-ws", item_id: "item-ws", delta: "hello" }, + { type: "response.output_item.added", response_id: "r-ws", output_index: 0, + item: { id: "item-ws", type: "function_call", call_id: "call-ws", name: "lookup", arguments: "{}" } }, + { type: "response.in_progress", response: { id: "r-ws", usage: { input_tokens: 4, output_tokens: 1 } } }, + ])("$type forbids HTTP replacement even with an unused grant", async event => { + installFake(ws => { + ws.emit("open", {}); + ws.emit("message", { data: JSON.stringify(event) }); + ws.emit("close", { code: 1006 }); + }); + const http = stubHttp(completed); + const budget = createRequestExecutionBudget(); + const response = await send(turn(), forwardConfig({ retryOnReset: {} }), undefined, budget); + // Depending on preflight, this is a projected failure or a body error. Neither + // representation may turn an observed semantic event into another inference. + await response.text().catch(() => ""); + expect(FakeWebSocket.instances).toHaveLength(1); + expect(FakeWebSocket.instances[0]!.sent).toHaveLength(1); + expect(http).toHaveLength(0); + expect(budget.claimAmbiguousResend?.(1)).toBe(true); + }); + + test("cancellation after send wins over a late socket death without spending a grant", async () => { + const abort = new AbortController(); + installFake(ws => { + ws.emit("open", {}); + abort.abort(); + ws.emit("close", { code: 1006 }); + }); + const http = stubHttp(completed); + const budget = createRequestExecutionBudget(); + const response = await send(turn(), forwardConfig({ retryOnReset: {} }), undefined, budget, abort.signal); + expect(response.status).toBe(499); + expect(FakeWebSocket.instances[0]!.sent).toHaveLength(1); + expect(http).toHaveLength(0); + expect(budget.claimAmbiguousResend?.(1)).toBe(true); + }); + + test("HTTP replacement preserves the sent request's model, input, tools and instructions", async () => { + installFake(SOCKET_DEATHS[0]![1]); + const http = stubHttp(completed); + const response = await send(turn({ + instructions: "Use the supplied lookup tool only when needed.", + input: [{ role: "user", content: "hello" }], + tools: [{ type: "function", name: "lookup", parameters: { type: "object", properties: {} } }], + }), forwardConfig({ retryOnReset: {} })); + await response.text(); + expect(http).toHaveLength(1); + const frame = JSON.parse(FakeWebSocket.instances[0]!.sent[0]!); + const replacement = JSON.parse(http[0]!); + for (const field of ["model", "input", "instructions", "tools", "store"]) { + expect(replacement[field]).toEqual(frame[field]); + } + }); + + test.each(["response.failed", "response.incomplete"])("HTTP %s remains terminal, not a third send", async type => { + installFake(SOCKET_DEATHS[0]![1]); + const http = stubHttp(() => new Response(`event: ${type}\ndata: ${JSON.stringify({ + type, + response: { id: "r-http", status: type.slice("response.".length), output: [], + ...(type === "response.failed" + ? { error: { code: "server_error", message: "failed" } } + : { incomplete_details: { reason: "max_output_tokens" } }) }, + })}\n\n`, { headers: { "content-type": "text/event-stream" } })); + const response = await send(turn(), forwardConfig({ retryOnReset: {} })); + await response.text().catch(() => ""); + expect(FakeWebSocket.instances).toHaveLength(1); + expect(http).toHaveLength(1); + }); + test.each([ ["a status the client would retry", () => new Response("busy", { status: 503 })], ["a reset of its own", () => { throw Object.assign(new Error("socket hang up"), { code: "ECONNRESET" }); }], From 462990242ab67fa6b1d0090b82d2bc07b3e06081 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:24:22 +0900 Subject: [PATCH 12/75] fix(desktop): keep a retry made between update attempts A retried install that finds the drain already settled clears wanted again without replacing the update snapshot, so a startup retry recorded only in wanted was lost when that install failed. resume() now promotes a pending snapshot as well. Found in the round 3 review of #6041. --- desktop/src-tauri/src/exit.rs | 37 ++++++++++++++++++++++++++++++++++- 1 file changed, 36 insertions(+), 1 deletion(-) diff --git a/desktop/src-tauri/src/exit.rs b/desktop/src-tauri/src/exit.rs index 5ee8e4bff10..8950c338638 100644 --- a/desktop/src-tauri/src/exit.rs +++ b/desktop/src-tauri/src/exit.rs @@ -323,7 +323,14 @@ impl ExitCoordinator { /// A person asked for a runtime again (the startup page's retry). pub fn resume(&self) { - self.inner().wanted = true; + let mut inner = self.inner(); + inner.wanted = true; + // A pending update snapshot must learn about the request too. A retried install that finds + // the drain already settled clears `wanted` again without replacing the snapshot, so a + // retry recorded only in `wanted` would be lost when that install fails and aborts. + if let Some(snapshot) = inner.restart_wanted.as_mut() { + *snapshot = true; + } } /// Reserve the right to start a runtime. False once something else owns the phase. @@ -803,6 +810,34 @@ mod tests { assert!(coordinator.supervision_allowed()); } + #[test] + fn a_retry_between_update_attempts_survives_the_second_claim() { + let coordinator = ExitCoordinator::new(); + coordinator.set_tray(TrayAvailability::Available); + assert!(coordinator.begin_stop()); + assert_eq!(coordinator.finish_stop(), None); + + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + Some(ExitReason::CoordinatedRestart) + ); + coordinator.finish_drain(DrainVerdict::Drained); + coordinator.resume(); + // The next install attempt finds the drain settled and clears `wanted` again. + assert_eq!( + coordinator.claim_drain(ExitReason::CoordinatedRestart), + None + ); + assert_eq!( + coordinator.abort_restart(), + Some(AbortedRestart { + phase: ExitPhase::Drained, + runtime_was_wanted: true, + }) + ); + assert!(coordinator.supervision_allowed()); + } + #[test] fn a_quit_or_a_drain_in_flight_is_never_aborted() { let coordinator = ExitCoordinator::new(); From bfbcbd6f8133678caeecaa342c3fbe12ea5c208b Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:24:22 +0900 Subject: [PATCH 13/75] fix(claude): refuse strict for a non-positive multipleOf --- src/adapters/anthropic-output-schema.ts | 3 +++ .../claude-integration/claude-output-schema-strict.test.ts | 6 ++++++ 2 files changed, 9 insertions(+) diff --git a/src/adapters/anthropic-output-schema.ts b/src/adapters/anthropic-output-schema.ts index bc1758b30e3..dccae46d211 100644 --- a/src/adapters/anthropic-output-schema.ts +++ b/src/adapters/anthropic-output-schema.ts @@ -225,6 +225,9 @@ export function satisfiesOpenAiStrictSchema(value: unknown, fineTuned = false): const constraint = node[key]; if (Object.hasOwn(node, key) && (typeof constraint !== "number" || !Number.isFinite(constraint))) return false; } + // JSON Schema requires a strictly positive divisor; a zero or negative one is a schema the + // destination rejects, so it cannot be certified strict. + if (Object.hasOwn(node, "multipleOf") && (node.multipleOf as number) <= 0) return false; for (const key of ["minItems", "maxItems"]) { const constraint = node[key]; if (Object.hasOwn(node, key) diff --git a/tests/claude-integration/claude-output-schema-strict.test.ts b/tests/claude-integration/claude-output-schema-strict.test.ts index f5ffc6dc635..ce045428038 100644 --- a/tests/claude-integration/claude-output-schema-strict.test.ts +++ b/tests/claude-integration/claude-output-schema-strict.test.ts @@ -139,6 +139,12 @@ describe("translated Anthropic structured output strict eligibility (#5901 follo }), true); }); + test.each([0, -2])("does not certify a non-positive multipleOf: %d", multipleOf => { + const schema = closedObject({ count: { type: "integer", multipleOf } }); + expect(satisfiesOpenAiStrictSchema(schema)).toBe(false); + expect(satisfiesOpenAiStrictSchema(closedObject({ count: { type: "integer", multipleOf: 2 } }))).toBe(true); + }); + test("fine-tuned targets preserve unsupported constraints with strict disabled", () => { const schema = closedObject({ nested: { From 2bfc18aca1ced5f5872284f943387621e446c5ba Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:24:22 +0900 Subject: [PATCH 14/75] docs(structure): record the executed ws-ambiguous-resend proof and its scope in ADR-4191 --- structure/decisions/ADR-4191-established-websocket-fallback.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/structure/decisions/ADR-4191-established-websocket-fallback.md b/structure/decisions/ADR-4191-established-websocket-fallback.md index 1dec5e87b9f..0e92ad3652d 100644 --- a/structure/decisions/ADR-4191-established-websocket-fallback.md +++ b/structure/decisions/ADR-4191-established-websocket-fallback.md @@ -11,4 +11,4 @@ - Chosen approach: Keep the existing dispatch-owned HTTP-only replacement. Require the provider's `retryOnReset` policy, self-contained request judgment, unspent request-wide grant and available send budget. Record and charge the physical send at the ordinary boundary, after credential selection is revalidated. Extend deterministic coverage rather than add a duplicate transport path. - Why: The exchange cannot independently authorize a new inference, reserve spend, refresh credentials or choose another account. The shared dispatch already owns those decisions and the request identity. Any semantic response event ends eligibility, including creation, tool and usage events before text. Quota control frames and pong events indicate liveness only. - Advantages, costs and consequences: One default replacement remains available without weakening hosted-tool, cancellation, timeout, native-control or committed-output exclusions. The opt-in still accepts possible duplicate inference billing; absence of output does not prove non-execution. An explicit `replacements: 2` retains the existing shared-policy contract for a subsequent HTTP reset, rather than becoming a second WebSocket fallback. This change does not widen that policy or alter runtime behavior. -- Validation boundary: Focused tests are authored but not executed in this work item because the operator prohibited test/build/install execution. Existing implementation and source traces are evidence of placement, not claims that the added assertions pass. +- Validation boundary: `tests/responses/ws-ambiguous-resend.test.ts` pins the behavior that landed in #5675 (`aed3bb8f42`); it runs in the hosted test shards. The fallback covers a socket that dies after the create frame and before the first Responses event; a death after output started stays a failed leg, because replaying it risks a second inference. From 91776636654d1b12dd4f66bbbc8c1c45af78c134 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:32:08 +0900 Subject: [PATCH 15/75] docs(devlog): close the 2.68.0 release round (#6060) --- .../000_plan.md | 0 .../001_app_and_external_evidence.md | 0 ...2_visualization_directive_normalization.md | 0 .../020_wp3_rebuild_and_live_verify.md | 0 .../030_done.md | 0 .../260927_release_2680/000_plan.md | 0 .../010_wp4_blockers_and_tray.md | 0 .../260927_release_2680/020_wp5_release.md | 45 +++++++++++++++++++ devlog/_fin/260927_release_2680/030_done.md | 38 ++++++++++++++++ .../260927_release_2680/020_wp5_release.md | 11 ----- 10 files changed, 83 insertions(+), 11 deletions(-) rename devlog/{_plan => _fin}/260927_directive_marker_bridge/000_plan.md (100%) rename devlog/{_plan => _fin}/260927_directive_marker_bridge/001_app_and_external_evidence.md (100%) rename devlog/{_plan => _fin}/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md (100%) rename devlog/{_plan => _fin}/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md (100%) rename devlog/{_plan => _fin}/260927_directive_marker_bridge/030_done.md (100%) rename devlog/{_plan => _fin}/260927_release_2680/000_plan.md (100%) rename devlog/{_plan => _fin}/260927_release_2680/010_wp4_blockers_and_tray.md (100%) create mode 100644 devlog/_fin/260927_release_2680/020_wp5_release.md create mode 100644 devlog/_fin/260927_release_2680/030_done.md delete mode 100644 devlog/_plan/260927_release_2680/020_wp5_release.md diff --git a/devlog/_plan/260927_directive_marker_bridge/000_plan.md b/devlog/_fin/260927_directive_marker_bridge/000_plan.md similarity index 100% rename from devlog/_plan/260927_directive_marker_bridge/000_plan.md rename to devlog/_fin/260927_directive_marker_bridge/000_plan.md diff --git a/devlog/_plan/260927_directive_marker_bridge/001_app_and_external_evidence.md b/devlog/_fin/260927_directive_marker_bridge/001_app_and_external_evidence.md similarity index 100% rename from devlog/_plan/260927_directive_marker_bridge/001_app_and_external_evidence.md rename to devlog/_fin/260927_directive_marker_bridge/001_app_and_external_evidence.md diff --git a/devlog/_plan/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md b/devlog/_fin/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md similarity index 100% rename from devlog/_plan/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md rename to devlog/_fin/260927_directive_marker_bridge/010_wp2_visualization_directive_normalization.md diff --git a/devlog/_plan/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md b/devlog/_fin/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md similarity index 100% rename from devlog/_plan/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md rename to devlog/_fin/260927_directive_marker_bridge/020_wp3_rebuild_and_live_verify.md diff --git a/devlog/_plan/260927_directive_marker_bridge/030_done.md b/devlog/_fin/260927_directive_marker_bridge/030_done.md similarity index 100% rename from devlog/_plan/260927_directive_marker_bridge/030_done.md rename to devlog/_fin/260927_directive_marker_bridge/030_done.md diff --git a/devlog/_plan/260927_release_2680/000_plan.md b/devlog/_fin/260927_release_2680/000_plan.md similarity index 100% rename from devlog/_plan/260927_release_2680/000_plan.md rename to devlog/_fin/260927_release_2680/000_plan.md diff --git a/devlog/_plan/260927_release_2680/010_wp4_blockers_and_tray.md b/devlog/_fin/260927_release_2680/010_wp4_blockers_and_tray.md similarity index 100% rename from devlog/_plan/260927_release_2680/010_wp4_blockers_and_tray.md rename to devlog/_fin/260927_release_2680/010_wp4_blockers_and_tray.md diff --git a/devlog/_fin/260927_release_2680/020_wp5_release.md b/devlog/_fin/260927_release_2680/020_wp5_release.md new file mode 100644 index 00000000000..f7e69660470 --- /dev/null +++ b/devlog/_fin/260927_release_2680/020_wp5_release.md @@ -0,0 +1,45 @@ +# 020 — wp5: release 2.68.0 + +Values for the 2.67.0 procedure: `CAND` = `origin/dev` after wp4 merges; `PV=2.68.0-preview.20260927`; +pre-move `dev-version-bump.yml --ref main -f intended-version=2.68.0 -f mode=pre-move` (dev → 2.69.0); +promotion branches `codex/260927-release-preview-2.68.0` and `codex/260927-release-main-2.68.0` built +with `git merge -s ours` and `scripts/release-version-sources.ts`; merge commits (never squash); push-event +CI and Service lifecycle at both promotion SHAs; `release.yml` preview first, then stable; verify npm +dist-tags, both GitHub releases' assets, and `latest.json` signatures; fast-forward local branches. + +Heuristic CI rule (owner): a failing job blocks only when it reproduces on rerun or its log points at a +change in main..dev. Runner-stall signatures get one job rerun. + +## wp5 P revalidation (2026-09-27) + +- `CAND=f764765c6453a718806d3465ea966015fa233123` (`origin/dev` after #6052). Lane=all run `36294278068` (workflow_dispatch) at CAND. +- `main` `4bc92294aa` (2.67.0, npm latest), `preview` `9c6fb1ee8b` (2.67.0-preview.20260926), `dev` 2.68.0. +- `PV=2.68.0-preview.20260927`. Pre-move: `gh workflow run dev-version-bump.yml --ref main -f intended-version=2.68.0 -f mode=pre-move` + (inputs verified on `origin/main`); its PR must change only the four version sources to 2.69.0. +- `release.yml` inputs verified: `version`, `tag`, `dry-run`, `resume-after-npm-publish`, `expected-sha`. + +## Audit (astra, NEAR-PASS) — folded + +- Order: the 2.69.0 pre-move PR merges before either release dispatch; `CAND` stays pinned before the bump. +- Both promotion branches start at `CAND` and merge their release branch with `-s ours`; `sync "$PV"` only on + preview, `check 2.68.0` on stable; promotion PRs merge with `--merge --match-head-commit`, never squash. +- The heuristic CI rule never waives release gates: push-event `ci.yml` and Service lifecycle must succeed at each + promotion merge SHA. +- Dispatch: `gh workflow run release.yml --ref preview -f version="$PV" -f tag=preview -f dry-run=false -f expected-sha=`; + stable only after the preview run succeeds, `--ref main -f version=2.68.0 -f tag=latest`. +- Verify: `npm view @bitkyc08/opencodex dist-tags --json`; `gh release view ` 25 assets, non-draft, prerelease flags; + `latest.json` 2.68.0 with five signatures. + +## wp5 execution log + +- Pre-move: `dev-version-bump.yml` run `36294309818` success opened #6053 (head `7af8381049`, four version sources + 2.68.0 → 2.69.0 only). Merged by admin as `99d0a9400e` under the owner's heuristic CI rule (workflow-token PR CI + does not start; the diff is the same four lines as every pre-move). +- Promotion: branches built in `/tmp/ocx-rel-2680` from CAND; `release-version-sources.ts check` passed for both; + main tree equals CAND, preview differs only in four version lines. #6054 `preview` merged as `09081803c5`, + #6055 `main` merged as `93f4231e4b` (merge commits, `--match-head-commit`). +- Runner hygiene: PR-event runs on the merged promotion branches and the superseded CAND lane=all run + `36294278068` were cancelled one at a time so the push-event runs on `main` and `preview` could start. + CAND's tree equals #6052's exact head, whose PR CI passed (31 pass, 6 skipped). +- Release gates pending: main CI `36294473376`, main Service lifecycle `36294473362`; preview CI `36294469680`, + preview Service lifecycle `36294469713`. diff --git a/devlog/_fin/260927_release_2680/030_done.md b/devlog/_fin/260927_release_2680/030_done.md new file mode 100644 index 00000000000..ef25119e2a9 --- /dev/null +++ b/devlog/_fin/260927_release_2680/030_done.md @@ -0,0 +1,38 @@ +# 030 — done: 2.68.0 release round + +## Outcome + +2.68.0 shipped from candidate `f764765c6453a718806d3465ea966015fa233123` as preview `2.68.0-preview.20260927` and +stable `2.68.0`. `dev` carries 2.69.0 (#6053). The main..dev regression review found three release blockers; all +were fixed in #6052 before the candidate was cut, together with the Windows/Linux tray parity the owner asked for. + +## Evidence + +- Regression review: seven astra lanes over main..dev (86 commits); dispositions in [000](000_plan.md) and [010](010_wp4_blockers_and_tray.md). +- #6052 exact-head PR CI: 31 pass, 6 skipped; merged as `f764765c64`. Its tree is the candidate. +- Promotion: #6054 `preview` `09081803c5`, #6055 `main` `93f4231e4b` (merge commits). +- Release-branch CI: preview Cross-platform CI `36294469680` (23 success, 5 skipped) and Service lifecycle `36294469713`; + main Cross-platform CI `36294473376` (23 success, 5 skipped) and Service lifecycle `36294473362`. +- Release runs: preview `36295546215` success (14/14), stable `36296387672` success (14/14). +- GitHub releases: `v2.68.0` (25 assets, prerelease false, target `93f4231e4b`) and `v2.68.0-preview.20260927` + (25 assets, prerelease true, target `09081803c5`). `latest.json`: 2.68.0 with signatures for darwin-aarch64, + darwin-x86_64, windows-x86_64, linux-x86_64 and linux-x86_64-deb. +- npm: `preview` = `2.68.0-preview.20260927` after registry propagation; the stable publish was acknowledged with + provenance and `latest` is checked again after propagation (the preview took about ten minutes). + +## Release-note items + +- Codex App inline visualizations work with every routed model (#6040, #6045). +- Windows/Linux tray: provider marks, 70%/90% quota colors, and switching the active account (#6052). +- Remote Link: Home-initiated links keep 2.67.0 forwarding; Child-initiated links require the tunnel ownership proof. +- Kiro: account model discovery, quota metrics, device login, concurrency caps, 1M GPT-5.6 context windows; failed + discovery backs off. +- Items listed by the review lanes in [000](000_plan.md) (combo cooldowns, stall defaults, desktop title strip, and others). + +## What did not go to plan + +- The first candidate lane=all run failed `test 2/4` on a batch timeout whose files all passed alone (runner stall); + it and the second candidate run were superseded or cancelled to free macOS runners. The candidate's tree was + covered by #6052's exact-head CI and by the push-event CI on both promotion commits. +- The pre-move PR (#6053) was merged without PR CI under the owner's heuristic rule; workflow-token PRs do not start CI. +- A 2.67.0-era Child join whose sidecar was deleted before its first 2.68.0 start still forwards like 2.67.0 (owner decision). diff --git a/devlog/_plan/260927_release_2680/020_wp5_release.md b/devlog/_plan/260927_release_2680/020_wp5_release.md deleted file mode 100644 index 2a79bda6dd5..00000000000 --- a/devlog/_plan/260927_release_2680/020_wp5_release.md +++ /dev/null @@ -1,11 +0,0 @@ -# 020 — wp5: release 2.68.0 - -Values for the 2.67.0 procedure: `CAND` = `origin/dev` after wp4 merges; `PV=2.68.0-preview.20260927`; -pre-move `dev-version-bump.yml --ref main -f intended-version=2.68.0 -f mode=pre-move` (dev → 2.69.0); -promotion branches `codex/260927-release-preview-2.68.0` and `codex/260927-release-main-2.68.0` built -with `git merge -s ours` and `scripts/release-version-sources.ts`; merge commits (never squash); push-event -CI and Service lifecycle at both promotion SHAs; `release.yml` preview first, then stable; verify npm -dist-tags, both GitHub releases' assets, and `latest.json` signatures; fast-forward local branches. - -Heuristic CI rule (owner): a failing job blocks only when it reproduces on rerun or its log points at a -change in main..dev. Runner-stall signatures get one job rerun. From bec870d6a8b5ca8986c3c2fddaa869f6e0f46b28 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:26:13 +0900 Subject: [PATCH 16/75] docs(devlog): record train 3 B1 review outcome and local proof --- .../_plan/260927_merge_train_3/010_batch1.md | 23 +++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/010_batch1.md b/devlog/_plan/260927_merge_train_3/010_batch1.md index 0a4ca773a5c..fca4bffba45 100644 --- a/devlog/_plan/260927_merge_train_3/010_batch1.md +++ b/devlog/_plan/260927_merge_train_3/010_batch1.md @@ -28,3 +28,26 @@ Captured to `.tmp/aside/` for every PR and issue above. All seven issues are own code pointers that match the PR premises. #4191's thread records the owner's position (Sep 21) that an established WebSocket dying mid-turn is a failed leg rather than an SSE fallback, plus two contributor data sets (Sep 23) asking for a fallback, so a test-and-ADR PR does not by itself resolve that issue. + +## Review outcome + +| PR | Kimi verdict | Folded | +|---|---|---| +| #6015 | LAND | — | +| #6034 | LAND; security review: no blocker | — | +| #6026 | LAND (cosmetic: `~` marker also shown for manual prices) | not folded; display only | +| #6019 | LAND-WITH-FIXES | `14913d98ad` pins the current user in the foreign-principal test | +| #6041 | LAND-WITH-FIXES | `462990242a` `resume()` updates a pending update snapshot; new two-attempt Rust test | +| #6006 | LAND-WITH-FIXES (optional) | `bfbcbd6f81` refuses strict for `multipleOf <= 0` | +| #6011 | LAND-WITH-FIXES | `2bfc18aca1` ADR-4191 validation bullet and scope | + +Accepted residuals: #6019 still compares owner and current user by record name, which the PR states and which predates +it; #6011 resolves #4191 only for socket death before the first Responses event. + +## Local proof at `2bfc18aca1` + +- `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan`: exit 0. +- Focused `bun test` on nine files: 207 pass, 4 skip, 0 fail. +- `cargo fmt --check` and `cargo test --lib -- exit:: updater::` (CI placeholder resources): 41 pass. + +Batch PR: #6059. From 8c7e05dd15536c7cc048ae853ea52f506d9c1ad9 Mon Sep 17 00:00:00 2001 From: luvs01 <27862058+luvs01@users.noreply.github.com> Date: Sun, 27 Sep 2026 14:35:52 +0900 Subject: [PATCH 17/75] test(compaction): drain fixture state and close routing history (#6057) Carried from #6057 into merge train round 3. Co-authored-by: luvs01 <27862058+luvs01@users.noreply.github.com> --- tests/helpers/compaction-routing-fixtures.ts | 38 ++++++++++++++++++- .../responses-compaction-routing.test.ts | 24 ++++++------ 2 files changed, 49 insertions(+), 13 deletions(-) diff --git a/tests/helpers/compaction-routing-fixtures.ts b/tests/helpers/compaction-routing-fixtures.ts index fb4db7b627b..1c2688692a0 100644 --- a/tests/helpers/compaction-routing-fixtures.ts +++ b/tests/helpers/compaction-routing-fixtures.ts @@ -1,13 +1,47 @@ import type { OcxConfig, OcxProviderConfig } from "../../src/types"; +import { afterAll, beforeAll } from "bun:test"; +import { clearResponseStateForTests, flushResponseState } from "../../src/responses/state"; +import { flushConfigDirHardeningForTests } from "../../src/config/paths"; +import { setAsyncIcaclsRunnerForTests, setIcaclsRunnerForTests } from "../../src/lib/windows-secret-acl"; +import { closeRequestHistoryIndex } from "../../src/routing/history/indexer"; +import { removeTreeWithRetry } from "./remove-tree"; /** * Config, request and upstream-response fixtures for the compaction-routing suite. * * Moved verbatim out of tests/responses/responses-compaction-routing.test.ts: that file sits at * its file-size cap, and the repository answer to a cap is a sibling helper rather than - * compressed control flow. Nothing here decides anything; every value is the one its callers - * were already building inline. + * compressed control flow. Fixture teardown also owns the continuation writes started by + * direct handler calls, so they cannot survive a fixture-home switch. */ +export function installCompactionRoutingAclFixture(): void { + // These cases prove routing and replay with synthetic credentials, not Windows DACLs. + // Actual ACL contracts have their own subprocess tests; incidental spawns here can + // outlive a case timeout and mutate the next fixture's continuation state. + beforeAll(() => { + const ok = { success: true, exitCode: 0, timedOut: false, stdout: "" }; + setIcaclsRunnerForTests(() => ok); + setAsyncIcaclsRunnerForTests(async () => ok); + }); + afterAll(async () => { + try { await flushConfigDirHardeningForTests(); } finally { + setIcaclsRunnerForTests(null); + setAsyncIcaclsRunnerForTests(null); + } + }); +} + +export async function drainCompactionResponseState(): Promise { + await flushResponseState(); + clearResponseStateForTests(); + closeRequestHistoryIndex(); +} + +export async function removeCompactionFixture(path: string): Promise { + await drainCompactionResponseState(); + removeTreeWithRetry(path); +} + export function keyProviderConfig(overrides: Partial = {}): OcxConfig { return { defaultProvider: "gw", diff --git a/tests/responses/responses-compaction-routing.test.ts b/tests/responses/responses-compaction-routing.test.ts index 52540f1dd4a..587159b58dd 100644 --- a/tests/responses/responses-compaction-routing.test.ts +++ b/tests/responses/responses-compaction-routing.test.ts @@ -41,12 +41,12 @@ import { acquireNativeMainProfileDrain, tryAdmitTurn } from "../../src/server/li import type { OcxConfig, OcxProviderConfig } from "../../src/types"; import { clearComboRecallForTests, recallComboForLane, rememberComboForLane } from "../../src/server/responses/combo-session-recall"; import { captureConfigGeneration } from "../../src/lib/state-store-sweeper"; -import { removeTreeWithRetry } from "../helpers/remove-tree"; -import { baseCompactionBody, compactionRequest, completedPayload, jsonResponse, keyProviderConfig, nativePoolConfig, sseResponse, twoAccountPoolConfig } from "../helpers/compaction-routing-fixtures"; +import { baseCompactionBody, compactionRequest, completedPayload, drainCompactionResponseState, installCompactionRoutingAclFixture, jsonResponse, keyProviderConfig, nativePoolConfig, removeCompactionFixture, sseResponse, twoAccountPoolConfig } from "../helpers/compaction-routing-fixtures"; import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; import { SERVER_BUDGET_MS } from "../helpers/test-budget"; const originalFetch = globalThis.fetch; +installCompactionRoutingAclFixture(); // A case that calls a handler directly never runs startServer, so it never takes the // spend-journal writer lease and its dispatch is refused before it reaches its own contract. @@ -56,9 +56,11 @@ let releaseSpendHome: (() => void) | undefined; const takeSpendHome = (): void => { releaseSpendHome ??= acquireOwnedSpendHome(); }; const dropSpendHome = (): void => { releaseSpendHome?.(); releaseSpendHome = undefined; }; -afterEach(() => { - dropSpendHome(); - globalThis.fetch = originalFetch; +afterEach(async () => { + try { await drainCompactionResponseState(); } finally { + dropSpendHome(); + globalThis.fetch = originalFetch; + } }); describe("supportsNativeResponsesCompactEndpoint (#422)", () => { @@ -323,7 +325,7 @@ describe("native compact usage reporting", () => { clearAccountQuota(); // Released before the directory holding it is removed. dropSpendHome(); - removeTreeWithRetry(testDir); + await removeCompactionFixture(testDir); if (previousOpencodexHome === undefined) delete process.env.OPENCODEX_HOME; else process.env.OPENCODEX_HOME = previousOpencodexHome; if (previousCodexHome === undefined) delete process.env.CODEX_HOME; @@ -398,7 +400,7 @@ describe("native Codex pool compaction", () => { clearCodexUpstreamHealth(); // Released before the directory holding it is removed. dropSpendHome(); - removeTreeWithRetry(testDir); + await removeCompactionFixture(testDir); if (previousOpencodexHome === undefined) delete process.env.OPENCODEX_HOME; else process.env.OPENCODEX_HOME = previousOpencodexHome; if (previousCodexHome === undefined) delete process.env.CODEX_HOME; @@ -473,7 +475,7 @@ describe("native Codex pool compaction", () => { clearCodexUpstreamHealth(); // Released before the directory holding it is removed. dropSpendHome(); - removeTreeWithRetry(testDir); + await removeCompactionFixture(testDir); if (previousOpencodexHome === undefined) delete process.env.OPENCODEX_HOME; else process.env.OPENCODEX_HOME = previousOpencodexHome; if (previousCodexHome === undefined) delete process.env.CODEX_HOME; @@ -543,7 +545,7 @@ describe("native Codex pool compaction", () => { clearCodexUpstreamHealth(); // Released before the directory holding it is removed. dropSpendHome(); - removeTreeWithRetry(testDir); + await removeCompactionFixture(testDir); if (previousOpencodexHome === undefined) delete process.env.OPENCODEX_HOME; else process.env.OPENCODEX_HOME = previousOpencodexHome; if (previousCodexHome === undefined) delete process.env.CODEX_HOME; @@ -886,14 +888,14 @@ describe("compact alternate-account attempt (#913)", () => { }); updateAccountQuota(id, id === "pool-a" ? 10 : 20); } - return run(twoAccountPoolConfig()).finally(() => { + return run(twoAccountPoolConfig()).finally(async () => { globalThis.fetch = originalFetch; clearCodexUpstreamHealth(); clearUpstreamHostHealth(); clearAccountQuota(); // Released before the directory holding it is removed. dropSpendHome(); - removeTreeWithRetry(testDir); + await removeCompactionFixture(testDir); if (previousOpencodexHome === undefined) delete process.env.OPENCODEX_HOME; else process.env.OPENCODEX_HOME = previousOpencodexHome; if (previousCodexHome === undefined) delete process.env.CODEX_HOME; From 766e58abcff127ec8ad441324b0c76d72def5ec8 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 14:35:53 +0900 Subject: [PATCH 18/75] fix(grok): account for metered preflight failures (#6035) Carried from #6035 into merge train round 3. Co-authored-by: Epinephrine --- src/server/responses/run-turn-execution.ts | 1 + structure/transports/responses-spend.md | 4 ++- .../responses-grok-devin-preflight.test.ts | 29 ++++++++++++++++++- 3 files changed, 32 insertions(+), 2 deletions(-) diff --git a/src/server/responses/run-turn-execution.ts b/src/server/responses/run-turn-execution.ts index fa1bb383323..de0dc848117 100644 --- a/src/server/responses/run-turn-execution.ts +++ b/src/server/responses/run-turn-execution.ts @@ -574,6 +574,7 @@ export async function executeResponsesRunTurn( if (preflight.replayUnsafe || preflight.error?.status !== 429 || preflight.error.code === SEND_BUDGET_EXHAUSTED_CODE) return; const { httpStatus, error } = adapterFailureFromEvent(preflight.error); + transportState.bindKeyUsageFromBridge(preflight.error.usage); cancelResponseCompletion(); runTurnAbort.abort(); queue.close(); diff --git a/structure/transports/responses-spend.md b/structure/transports/responses-spend.md index 8f81cfe15a3..6bfc03a8938 100644 --- a/structure/transports/responses-spend.md +++ b/structure/transports/responses-spend.md @@ -109,7 +109,9 @@ during this process's lifetime can still be released for free. Settlement follows what the request learned. The terminal usage belongs to the last send that left, so that one settles with the real figure; every earlier send failed without reporting usage of its own and may still have been billed, so it becomes unresolved spend rather than free. A -request that reports no usage at all leaves all of them unresolved. If deferred settlement reaches a tracker with reserved sends after its ledger lease ends, only `SPEND_LEDGER_OWNER_NOT_HELD` is dropped with the discarded ledger. Other owner and storage failures propagate with pending send IDs intact so settlement can be retried. +request that reports no usage at all leaves all of them unresolved. A pre-output Grok/Devin 429 +binds usage carried by its error event before returning the HTTP refusal, just as the ordinary +streaming and buffered bridges bind terminal usage. If deferred settlement reaches a tracker with reserved sends after its ledger lease ends, only `SPEND_LEDGER_OWNER_NOT_HELD` is dropped with the discarded ledger. Other owner and storage failures propagate with pending send IDs intact so settlement can be retried. Replay resolves what nobody is left to settle, and resolves it as unresolved spend whatever state it was in. Giving an undispatched one its tokens back would assume the journal is complete up to diff --git a/tests/responses/responses-grok-devin-preflight.test.ts b/tests/responses/responses-grok-devin-preflight.test.ts index 78d3bc5740d..f291ddfd131 100644 --- a/tests/responses/responses-grok-devin-preflight.test.ts +++ b/tests/responses/responses-grok-devin-preflight.test.ts @@ -1,4 +1,6 @@ import { afterAll, afterEach, beforeEach, expect, mock, test } from "bun:test"; +import { existsSync, readFileSync } from "node:fs"; +import { join } from "node:path"; import type { ProviderAdapter } from "../../src/adapters/base"; import type { AdapterEvent, OcxConfig, OcxProviderConfig } from "../../src/types"; import { saveCredential } from "../../src/oauth/store"; @@ -34,6 +36,8 @@ mock.module("../../src/server/adapter-resolve", () => ({ ...resolver, }, })); const { handleResponses } = await import("../../src/server/responses"); +const { addFinalRequestLog } = await import("../../src/server/request-log"); +let requestLog: Parameters[2]; let home: ReturnType; let release: (() => void) | undefined; beforeEach(async () => { @@ -73,10 +77,11 @@ function run({ stallTimeoutSec, providers: { devin: { adapter: "devin", authMode: "oauth", baseUrl: "https://server.codeium.com", models: ["swe-2"] } }, } as OcxConfig; + requestLog = { model: "", provider: "", surface }; return handleResponses(new Request("http://localhost/v1/responses", { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ model: "devin/swe-2", input: "answer", stream }), - }), config, { model: "", provider: "", surface }, { comboAttempt, abortSignal }); + }), config, requestLog, { comboAttempt, abortSignal }); } async function waitForPreflightResponse(pending: Promise, started: Promise, resume: () => void) { @@ -113,6 +118,28 @@ test.each([ expect(calls).toBe(1); }); +test.each([true, false])("pre-output 429 preserves metered usage (stream=%s)", async stream => { + const usage = { inputTokens: 23, outputTokens: 5, totalTokens: 28 }; + events = [{ ...limit, usage }]; + expect((await run({ stream })).status).toBe(429); + expect(requestLog.usage).toEqual(usage); +}); + +test.each([true, false])("pre-output 429 persists bound usage in the usage journal (stream=%s)", async stream => { + const usage = { inputTokens: 23, outputTokens: 5, totalTokens: 28 }; + events = [{ ...limit, usage }]; + expect((await run({ stream })).status).toBe(429); + // Exercise usage-journal finalization with this request's bound metering context. + // The durable row proves usage survives logging, not that a separate API-key budget + // reservation was settled. This preflight refusal returns HTTP before starting SSE. + addFinalRequestLog(`preflight-usage-${stream}`, Date.now(), requestLog, 429, { closeReason: "non_stream" }); + const journalPath = join(home.root, "usage.jsonl"); + expect(existsSync(journalPath)).toBe(true); + const rows = readFileSync(journalPath, "utf8").trim().split("\n").filter(Boolean) + .map(line => JSON.parse(line) as { usage?: typeof usage }); + expect(rows.at(-1)?.usage).toMatchObject({ inputTokens: 23, outputTokens: 5 }); +}); + test("buffered cooldown heartbeat keeps a final refusal as HTTP 429", async () => { events = [{ type: "heartbeat", preflightReady: true }, limit]; From 429fd60ef13d2c3937f971dbb03876fdb8724911 Mon Sep 17 00:00:00 2001 From: mdwsk88 <924038395@qq.com> Date: Sun, 27 Sep 2026 14:35:55 +0900 Subject: [PATCH 19/75] fix(codebuddy): harden captured parallel tool blocks (#6022) Carried from #6022 into merge train round 3. Co-authored-by: mdwsk88 <924038395@qq.com> --- .../src/content/docs/guides/providers.md | 2 +- src/adapters/coding-agent/protocol.ts | 48 +++++++++-- src/adapters/coding-agent/turn.ts | 37 ++++---- structure/providers-and-adapters.md | 8 +- tests/providers/codebuddy-protocol.test.ts | 51 +++++++++++ .../codebuddy-tool-bridge-turn.test.ts | 86 +++++++++++++++++++ 6 files changed, 204 insertions(+), 28 deletions(-) diff --git a/docs-site/src/content/docs/guides/providers.md b/docs-site/src/content/docs/guides/providers.md index 1d51fc11d2a..6a91e7fe200 100644 --- a/docs-site/src/content/docs/guides/providers.md +++ b/docs-site/src/content/docs/guides/providers.md @@ -907,7 +907,7 @@ OpenCodex provides official adapter support for Tencent Cloud's CodeBuddy Code C - **Region Isolation:** `codebuddy` and `codebuddy-cn` use separate canonical endpoints (`https://www.codebuddy.ai` and `https://www.codebuddy.cn`) and isolated child environments (`CODEBUDDY_INTERNET_ENVIRONMENT=public` vs `internal`). Credentials are strictly region-scoped and never exchanged across environments. Overriding the canonical base URL fails closed. - **Model Discovery:** the proxy requests the CodeBuddy product configuration (`GET {baseUrl}/v3/config`) with the configured key as the `X-API-Key` header, and the roster in that answer is the authoritative roster of discovered models: it is the key's own account configuration, so it is proven to belong to the key — a different or wrong key answers the anonymous envelope with no roster instead of another account's models. The authenticated roster is the same list the CLI prints for `--model` (the "Currently supported" line of a signed-in CLI), can differ from the static manifest bundled with the CLI, and the vendor default selectors (`default` for CN, `default-model` for Global) never appear in it but remain callable: the catalog retains them during live discovery and on every fallback path. On start/sync the proxy binds the cached roster to an irreversible fingerprint of the configured key, so a key switch never observes a roster cached for the previous key, and degrades to the stale provider/key-fingerprint-scoped cache, then to the static seed in `src/providers/codebuddy-models.ts`, when the key does not authenticate or the request fails. Discovery failure logs contain only a category and HTTP status, without the gateway's message or a raw transport exception. The credentialed discovery request does not follow redirects; a 3xx response degrades the roster without forwarding the key to another origin. -- **Tool Ownership and the Tool Bridge:** The CLI is always spawned with `--tools ""` and `--strict-mcp-config`, so it has no built-in or user-configured tools of its own. When a request carries a Codex tool catalog, the provider arms a capture-only MCP bridge: the validated catalog and MCP config are written to a private temp dir, the CLI is launched with `--mcp-config` and an exact `--allowedTools` list, and the `system/init` frame must report exactly that bridge server as connected or the turn fails closed. The bridge advertises the Codex tools and captures proposed calls but never executes anything: a completed tool-call batch is returned as `function_call` items (names mapped back to the request's wire names, at most 16 calls per assistant message), the process tree is terminated at `message_stop`, and the external Codex client alone performs approval, sandboxing, and execution. Tool results come back as the next request's input, and the conversation continues. Requests without tools keep the plain text-and-reasoning shape. If the CLI writes an unquoted DSML `calls` control line followed by a `functions.*` invoke control line into text or reasoning, OpenCodex refuses the turn instead of forwarding the scaffold or interpreting it as an executable call. DSML discussed or quoted in prose, inline code, fenced code, or source examples remains ordinary answer text. +- **Tool Ownership and the Tool Bridge:** The CLI is always spawned with `--tools ""` and `--strict-mcp-config`, so it has no built-in or user-configured tools of its own. When a request carries a Codex tool catalog, the provider arms a capture-only MCP bridge: the validated catalog and MCP config are written to a private temp dir, the CLI is launched with `--mcp-config` and an exact `--allowedTools` list, and the `system/init` frame must report exactly that bridge server as connected or the turn fails closed. The bridge advertises the Codex tools and captures proposed calls but never executes anything: a completed tool-call batch is returned as `function_call` items (names mapped back to the request's wire names, at most 16 calls per assistant message), the process tree is terminated at `message_stop`, and the external Codex client alone performs approval, sandboxing, and execution. Parallel tool-use blocks are buffered and returned as complete calls even when their fragments interleave or CodeBuddy reuses a content-block index. A block left open at turn end, a same-index replacement before complete JSON arguments, or an argument fragment that cannot be attributed to an open block fails the turn instead of forwarding a partial or empty call. Tool results come back as the next request's input, and the conversation continues. Requests without tools keep the plain text-and-reasoning shape. If the CLI writes an unquoted DSML `calls` control line followed by a `functions.*` invoke control line into text or reasoning, OpenCodex refuses the turn instead of forwarding the scaffold or interpreting it as an executable call. DSML discussed or quoted in prose, inline code, fenced code, or source examples remains ordinary answer text. - **Tool Choice Enforcement:** When a request specifies `tool_choice: "required"` or selects a specific named tool, the bridge expects a tool call from the model. If the CLI completes the turn with plain text instead of capturing a tool call, OpenCodex fails closed with a 502 `tool_call_required` error rather than returning an invalid text completion. - **Governance Status:** Whether routing this vendor automation surface behind a proxy for a third-party agent satisfies CodeBuddy's acceptable-use terms is an open question flagged for maintainer security review (see the governance note in the provider registry entry). Treat this provider as pending that review, and keep the tool bridge's ownership boundary in mind: the nested CLI advertises tools but never executes them, and approval, sandboxing, and execution remain with the external Codex client. - **Entitlements and Billing:** The provider uses the same vendor-documented CodeBuddy account/CLI authentication surface. Availability and billing of free, promotional, trial, or subscription credits remain determined by the user's CodeBuddy account entitlement. diff --git a/src/adapters/coding-agent/protocol.ts b/src/adapters/coding-agent/protocol.ts index 71085e30f47..43c76ea29fb 100644 --- a/src/adapters/coding-agent/protocol.ts +++ b/src/adapters/coding-agent/protocol.ts @@ -241,6 +241,8 @@ export interface StreamParseState { toolBlockStarts?: number; /** Completed tool_use content blocks observed in this stream. */ completedToolCalls?: number; + /** CodeBuddy's capture-only bridge requires complete JSON and matching block indices. */ + strictToolBlockCapture?: boolean; /** Tool IDs already captured through partial events, for complete-assistant deduplication. */ partialToolCallIds?: Set; /** A complete assistant tool block had no matching partial capture. */ @@ -351,6 +353,7 @@ export interface OpenToolBlock { id: string; name: string; argParts: string[]; + indexed: boolean; } /** Key a tool_use start frame by content-block index, falling back to a synthetic key. */ @@ -366,8 +369,9 @@ function toolBlockKey(state: StreamParseState, event: StreamMessage): number { * Resolve a delta/stop frame to an open tool block. An indexed frame only matches a block * opened under the same index — CodeBuddy skips stop frames for thinking blocks, and such a * stop must not close a tool block that happens to be open. An index-less frame resolves - * only when exactly one block is open. Ambiguous argument deltas are rejected before - * resolution; an unmatched stop cannot close a tool block. + * only when exactly one block is open. Ambiguous argument deltas with multiple open blocks + * fail the turn. On the CodeBuddy capture path, a missing-index delta or stop cannot match + * an indexed block; an unmatched stop leaves the block open for terminal accounting. */ function resolveToolBlockKey(state: StreamParseState, event: StreamMessage): number | undefined { const index = event.index; @@ -376,16 +380,33 @@ function resolveToolBlockKey(state: StreamParseState, event: StreamMessage): num } const blocks = state.openToolBlocks; if (!blocks || blocks.size !== 1) return undefined; - return blocks.keys().next().value; + const [key, block] = blocks.entries().next().value!; + return state.strictToolBlockCapture && block.indexed ? undefined : key; } /** * Emit a closed block atomically — start, the buffered fragments in arrival order, end — * so the strictly sequential downstream bridge never sees two calls open at once. */ -function closeToolBlock(state: StreamParseState, key: number, events: AdapterEvent[]): void { +function closeToolBlock(state: StreamParseState, key: number, events: AdapterEvent[], implicit = false): void { const block = state.openToolBlocks?.get(key); if (!block || !state.openToolBlocks) return; + if (state.strictToolBlockCapture && (implicit || block.argParts.length > 0)) { + // A second start on this index is an implicit stop only when the previous call's + // arguments are already complete. Otherwise a later delta could be assigned to the + // wrong call and still produce a superficially successful tool-use turn. An explicit + // stop with no deltas retains the CLI's existing empty-arguments representation. + const argumentsJson = block.argParts.join(""); + let parsedArguments: unknown; + try { + parsedArguments = JSON.parse(argumentsJson); + } catch { + throw new CodingAgentProtocolError("Coding-agent CLI ended a tool call with incomplete JSON arguments."); + } + if (!asRecord(parsedArguments)) { + throw new CodingAgentProtocolError("Coding-agent CLI ended a tool call with non-object JSON arguments."); + } + } state.openToolBlocks.delete(key); events.push({ type: "tool_call_start", id: block.id, name: block.name }); for (const part of block.argParts) events.push({ type: "tool_call_delta", arguments: part }); @@ -428,6 +449,11 @@ function mapRawStreamEvent(event: StreamMessage, state: StreamParseState): Adapt if (partial) { const key = resolveToolBlockKey(state, event); const block = key === undefined ? undefined : state.openToolBlocks?.get(key); + if (state.strictToolBlockCapture && !block) { + throw new CodingAgentProtocolError( + "Coding-agent CLI sent a tool argument delta that cannot be attributed to an open tool block.", + ); + } if (block) block.argParts.push(partial); } } @@ -445,12 +471,16 @@ function mapRawStreamEvent(event: StreamMessage, state: StreamParseState): Adapt // CodeBuddy reuses one content-block index for a parallel batch: every call in the // batch starts on the same index, intermediate blocks never receive a stop, and only // the final block does (observed 2026-09-26: START 2 alpha, A's complete args, START 2 - // beta, B's complete args, one STOP 2). Parallel calls stream their arguments - // sequentially — never interleaved — so the block already open on this index is - // complete, and the new start implicitly closes it. - closeToolBlock(state, key, events); + // beta, B's complete args, one STOP 2). A new start implicitly closes the previous + // block only after the capture path verifies its arguments form a complete object. + closeToolBlock(state, key, events, true); } - (state.openToolBlocks ??= new Map()).set(key, { id, name, argParts: [] }); + (state.openToolBlocks ??= new Map()).set(key, { + id, + name, + argParts: [], + indexed: typeof event.index === "number" && Number.isInteger(event.index), + }); state.toolBlockStarts = (state.toolBlockStarts ?? 0) + 1; state.partialToolCallIds?.add(id); } diff --git a/src/adapters/coding-agent/turn.ts b/src/adapters/coding-agent/turn.ts index af9c6de946e..ba731b701d5 100644 --- a/src/adapters/coding-agent/turn.ts +++ b/src/adapters/coding-agent/turn.ts @@ -381,6 +381,7 @@ export async function runCodingAgentTurn(input: CodingAgentTurnInput): Promise() : undefined, }; @@ -459,6 +460,25 @@ export async function runCodingAgentTurn(input: CodingAgentTurnInput): Promise toolCallStarts) { + // The parser buffers a block until its stop (or same-index replacement), so the + // per-turn limit must be checked when the block opens, not when its buffered + // tool_call_start is finally emitted. The init handshake is already gated above. + toolCallStarts = state.toolBlockStarts!; + if (toolCallStarts > toolBridge.maxTurnToolCalls) { + emitOnce({ + type: "error", + message: `Coding-agent CLI returned more than the ${toolBridge.maxTurnToolCalls}-tool-call turn limit.`, + status: 502, + errorType: "upstream_error", + code: "tool_call_limit", + retryable: false, + }); + failClosed = true; + kill(); + break; + } + } for (const event of mappedEvents) { if (toolBridge && !initValidated && event.type === "done") { emitOnce({ @@ -474,20 +494,6 @@ export async function runCodingAgentTurn(input: CodingAgentTurnInput): Promise toolBridge.maxTurnToolCalls) { - emitOnce({ - type: "error", - message: `Coding-agent CLI returned more than the ${toolBridge.maxTurnToolCalls}-tool-call turn limit.`, - status: 502, - errorType: "upstream_error", - code: "tool_call_limit", - retryable: false, - }); - failClosed = true; - kill(); - break; - } const wireName = toolBridge.emittedNameMap.get(event.name); if (wireName === undefined) { emitOnce({ @@ -593,8 +599,7 @@ export async function runCodingAgentTurn(input: CodingAgentTurnInput): Promise 0) { - // Raw tool_use starts are gated before buffering, so every completed call was admitted - // after the bridge init handshake. + // The raw-start gate above already refused any call opened before the handshake. // The capture-only MCP handler never answers, so the CLI parks after message_stop. // The completed tool_use blocks are this turn's structured output: end the leg here // and terminate the tree; the client executes, and the next request continues. diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 8f262d4dc1a..511f0ce8d26 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -1,8 +1,12 @@ # Providers And Adapters The coding-agent stream parser buffers each tool-use block by its content-block index -and emits a complete start/delta/end sequence on closure. A new start on an occupied -index closes the previous block; distinct indices can interleave. Turn completion +and emits a complete start/delta/end sequence on closure. Distinct indices can interleave. +For the CodeBuddy capture-only bridge, the init handshake and turn-call limit are checked +when a block opens, before its buffered events can be emitted. A new start on an occupied +index closes the previous block only when its arguments form a complete JSON object; +an unindexed delta or stop cannot be attributed to an indexed block, and a nonempty +argument delta that cannot be attributed fails immediately. Turn completion requires every opened block to close, preserving the downstream single-open-call contract. An indexless argument delta belongs to the sole open block; with multiple blocks open, the parser fails the turn before releasing their buffered calls. diff --git a/tests/providers/codebuddy-protocol.test.ts b/tests/providers/codebuddy-protocol.test.ts index ffafe2966c4..94f33c5c2c0 100644 --- a/tests/providers/codebuddy-protocol.test.ts +++ b/tests/providers/codebuddy-protocol.test.ts @@ -7,6 +7,7 @@ import { mapStreamMessageToEvents, projectedHistoryCharLimit, readJsonLines, + type StreamParseState, usageFromResult, } from "../../src/adapters/coding-agent/protocol"; import type { OcxParsedRequest } from "../../src/types"; @@ -311,6 +312,56 @@ describe("codebuddy stream-json event mapping", () => { expect(state.openToolBlocks?.size ?? 0).toBe(0); }); + test.each(["", "{\"value\":"])("same-index reuse rejects incomplete arguments %j before closing the previous call", partial => { + const state: StreamParseState = { + sawPartialText: false, + sawPartialThinking: false, + sawTerminalResult: false, + strictToolBlockCapture: true, + }; + const feed = (event: unknown) => mapStreamMessageToEvents({ type: "stream_event", event: event as Record }, state); + feed({ type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_a", name: "alpha" } }); + if (partial) feed({ type: "content_block_delta", index: 2, delta: { type: "input_json_delta", partial_json: partial } }); + + expect(() => feed({ type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_b", name: "beta" } })) + .toThrow("incomplete JSON arguments"); + expect(state.completedToolCalls ?? 0).toBe(0); + expect(state.openToolBlocks?.get(2)?.id).toBe("tu_a"); + }); + + test("same-index reuse rejects complete JSON that is not an argument object", () => { + const state: StreamParseState = { + sawPartialText: false, + sawPartialThinking: false, + sawTerminalResult: false, + strictToolBlockCapture: true, + }; + const feed = (event: unknown) => mapStreamMessageToEvents({ type: "stream_event", event: event as Record }, state); + feed({ type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_a", name: "alpha" } }); + feed({ type: "content_block_delta", index: 2, delta: { type: "input_json_delta", partial_json: "[]" } }); + + expect(() => feed({ type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_b", name: "beta" } })) + .toThrow("non-object JSON arguments"); + expect(state.completedToolCalls ?? 0).toBe(0); + expect(state.openToolBlocks?.get(2)?.id).toBe("tu_a"); + }); + + test("an unindexed argument delta cannot be dropped from the sole indexed CodeBuddy tool block", () => { + const state: StreamParseState = { + sawPartialText: false, + sawPartialThinking: false, + sawTerminalResult: false, + strictToolBlockCapture: true, + }; + const feed = (event: unknown) => mapStreamMessageToEvents({ type: "stream_event", event: event as Record }, state); + feed({ type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_a", name: "alpha" } }); + + expect(() => feed({ type: "content_block_delta", delta: { type: "input_json_delta", partial_json: "{\"wrong\":true}" } })) + .toThrow("tool argument delta that cannot be attributed"); + expect(state.openToolBlocks?.get(2)?.argParts).toEqual([]); + expect(state.completedToolCalls ?? 0).toBe(0); + }); + test("usageFromResult returns undefined when no usage is present", () => { expect(usageFromResult({ type: "result" })).toBeUndefined(); }); diff --git a/tests/providers/codebuddy-tool-bridge-turn.test.ts b/tests/providers/codebuddy-tool-bridge-turn.test.ts index 98598f8f790..51df4d0d89d 100644 --- a/tests/providers/codebuddy-tool-bridge-turn.test.ts +++ b/tests/providers/codebuddy-tool-bridge-turn.test.ts @@ -417,6 +417,28 @@ describe("CodeBuddy capture-only tool bridge turn", () => { expect(child?.killed).toBe(true); }); + test("an indexless argument delta for the sole indexed tool block fails before a later indexed stop", async () => { + const p = parsed([tool("exec")]); + const cliName = [...buildCodeBuddyToolBridge(p).emittedNameMap.keys()][0]!; + let child: FakeChild | undefined; + const spawn: SpawnFn = () => { + child = fakeChild(frameLines([ + INIT_OK, + { type: "stream_event", event: { type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_a", name: cliName } } }, + inputJsonDelta("{\"wrong\":true}"), + { type: "stream_event", event: { type: "content_block_stop", index: 2 } }, + MESSAGE_STOP, + ])); + return child as unknown as ChildProcess; + }; + const adapter = createCodeBuddyAdapter(provider(), { spawn, which: () => "/usr/bin/codebuddy" }); + const events = await run(adapter, p); + + expect(events).toEqual([expect.objectContaining({ type: "error", code: "protocol_error", status: 502, retryable: false })]); + expect(events.some(e => e.type === "tool_call_start" || e.type === "tool_call_delta" || e.type === "done")).toBe(false); + expect(child?.killed).toBe(true); + }); + test("a parallel batch on one shared block index completes every call in the leg", async () => { const p = parsed([tool("exec")]); const bridge = buildCodeBuddyToolBridge(p); @@ -465,6 +487,48 @@ describe("CodeBuddy capture-only tool bridge turn", () => { expect(child?.killed).toBe(true); }); + test.each(["", "{\"command\":"])("same-index reuse fails closed when the previous arguments are incomplete %j", async partial => { + const p = parsed([tool("exec")]); + const cliName = [...buildCodeBuddyToolBridge(p).emittedNameMap.keys()][0]!; + const start = (id: string) => ({ + type: "stream_event", + event: { type: "content_block_start", index: 2, content_block: { type: "tool_use", id, name: cliName } }, + }); + const frames: unknown[] = [INIT_OK, start("tu_a")]; + if (partial) frames.push({ + type: "stream_event", + event: { type: "content_block_delta", index: 2, delta: { type: "input_json_delta", partial_json: partial } }, + }); + frames.push(start("tu_b"), MESSAGE_STOP); + const adapter = createCodeBuddyAdapter(provider(), { + spawn: () => fakeChild(frameLines(frames)) as unknown as ChildProcess, + which: () => "/usr/bin/codebuddy", + }); + + const events = await run(adapter, p); + expect(events.at(-1)).toMatchObject({ type: "error", code: "protocol_error", status: 502, retryable: false }); + expect(events.some(e => e.type === "tool_call_start" || e.type === "done")).toBe(false); + }); + + test("an unindexed stop cannot complete an indexed tool call", async () => { + const p = parsed([tool("exec")]); + const cliName = [...buildCodeBuddyToolBridge(p).emittedNameMap.keys()][0]!; + const adapter = createCodeBuddyAdapter(provider(), { + spawn: () => fakeChild(frameLines([ + INIT_OK, + { type: "stream_event", event: { type: "content_block_start", index: 2, content_block: { type: "tool_use", id: "tu_a", name: cliName } } }, + { type: "stream_event", event: { type: "content_block_delta", index: 2, delta: { type: "input_json_delta", partial_json: "{}" } } }, + BLOCK_STOP, + MESSAGE_STOP, + ])) as unknown as ChildProcess, + which: () => "/usr/bin/codebuddy", + }); + + const events = await run(adapter, p); + expect(events.at(-1)).toMatchObject({ type: "error", code: "protocol_error", status: 502, retryable: false }); + expect(events.some(e => e.type === "tool_call_start" || e.type === "done")).toBe(false); + }); + test("a tool call before the init frame fails closed with tool_bridge_init_missing", async () => { const p = parsed([tool("exec")]); const bridge = buildCodeBuddyToolBridge(p); @@ -741,6 +805,7 @@ describe("CodeBuddy capture-only tool bridge turn", () => { const frames: unknown[] = [INIT_OK]; for (let i = 0; i < 17; i += 1) { frames.push(toolUseStart(cliName, `tu_${i}`)); + frames.push(inputJsonDelta("{}")); frames.push(BLOCK_STOP); } frames.push(MESSAGE_STOP); @@ -751,4 +816,25 @@ describe("CodeBuddy capture-only tool bridge turn", () => { const events = await run(adapter, p); expect(events.at(-1)).toMatchObject({ type: "error", code: "tool_call_limit" }); }); + + test("the turn limit is enforced when the seventeenth block opens", async () => { + const p = parsed([tool("exec")]); + const cliName = [...buildCodeBuddyToolBridge(p).emittedNameMap.keys()][0]!; + const frames: unknown[] = [INIT_OK]; + for (let i = 0; i < 17; i += 1) { + frames.push({ + type: "stream_event", + event: { type: "content_block_start", index: i, content_block: { type: "tool_use", id: `tu_${i}`, name: cliName } }, + }); + } + frames.push(MESSAGE_STOP); + const adapter = createCodeBuddyAdapter(provider(), { + spawn: () => fakeChild(frameLines(frames)) as unknown as ChildProcess, + which: () => "/usr/bin/codebuddy", + }); + + const events = await run(adapter, p); + expect(events.at(-1)).toMatchObject({ type: "error", code: "tool_call_limit", status: 502, retryable: false }); + expect(events.some(e => e.type === "tool_call_start" || e.type === "done")).toBe(false); + }); }); From 4026640b78b9d690bbfa21e6ceef9739db5ea798 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 14:35:56 +0900 Subject: [PATCH 20/75] fix(web-search): release sidecar probes on rejection and response settlement (#6047) Carried from #6047 into merge train round 3. Co-authored-by: Epinephrine --- src/server/responses/core.ts | 1 + src/server/responses/sidecar-execution.ts | 17 ++- .../responses-run-turn-web-search.test.ts | 144 +++++++++++++++++- 3 files changed, 159 insertions(+), 3 deletions(-) diff --git a/src/server/responses/core.ts b/src/server/responses/core.ts index a1ad4a15b37..0f231ad86e3 100644 --- a/src/server/responses/core.ts +++ b/src/server/responses/core.ts @@ -140,6 +140,7 @@ async function handleResponsesInner( responseEffects, sendBudgetState, ); + if (sidecarPlans instanceof Response && !sidecarPlans.ok) sidecarState.openAiSidecar?.releaseProbeLease?.(); if (sidecarPlans instanceof Response) return sidecarPlans; const completionPolicy = createResponsesCompletionPolicy(requestContext, sidecarState); if (transportState.adapter.runTurn) return await executeResponsesRunTurn( diff --git a/src/server/responses/sidecar-execution.ts b/src/server/responses/sidecar-execution.ts index a369a903bef..a0cb6ac07d5 100644 --- a/src/server/responses/sidecar-execution.ts +++ b/src/server/responses/sidecar-execution.ts @@ -111,6 +111,17 @@ export async function executeResponsesSidecars( cancelResponseCompletion, } = responseEffects; + // Resolving the OpenAI search credential may hold the account's sole + // cooldown-recovery probe lease. A streamed sidecar result keeps it until the + // stream settles — completion or client cancel — so a later in-stream search + // outcome can still clear the cooldown; a response with no live body is + // terminal, so the lease is handed back before returning it. A recorded + // search outcome already settled the lease, making each release a + // generation-bound no-op. + const releaseSearchProbeLease = (): void => { + openAiSidecar?.releaseProbeLease?.(); + }; + // Tool results are PAIRED by call_id. parseRequest writes it into OcxToolResultMessage.toolCallId // (parser.ts:738/752) without validating it, because inputItemSchema's permissive catch-all @@ -431,11 +442,12 @@ export async function executeResponsesSidecars( if (imgResponse.body) { const imgTurnAc = new AbortController(); imgTurnAc.signal.addEventListener("abort", cancelResponseCompletion, { once: true }); - return new Response(trackStreamLifetime(imgResponse.body, imgTurnAc, undefined, options.turnAdmissionLease), { + return new Response(trackStreamLifetime(imgResponse.body, imgTurnAc, releaseSearchProbeLease, options.turnAdmissionLease), { status: imgResponse.status, headers: imgResponse.headers, }); } + releaseSearchProbeLease(); return imgResponse; } // end else (streaming bridge) } @@ -511,11 +523,12 @@ export async function executeResponsesSidecars( if (wsResponse.body) { const wsTurnAc = new AbortController(); wsTurnAc.signal.addEventListener("abort", cancelResponseCompletion, { once: true }); - return new Response(trackStreamLifetime(wsResponse.body, wsTurnAc, undefined, options.turnAdmissionLease), { + return new Response(trackStreamLifetime(wsResponse.body, wsTurnAc, releaseSearchProbeLease, options.turnAdmissionLease), { status: wsResponse.status, headers: wsResponse.headers, }); } + releaseSearchProbeLease(); return wsResponse; } diff --git a/tests/responses/responses-run-turn-web-search.test.ts b/tests/responses/responses-run-turn-web-search.test.ts index 1477290f7ab..afbf1d50c49 100644 --- a/tests/responses/responses-run-turn-web-search.test.ts +++ b/tests/responses/responses-run-turn-web-search.test.ts @@ -28,9 +28,23 @@ function fixture(provider: OcxProviderConfig): ProviderAdapter { }, }; } +function fetchFixture(provider: OcxProviderConfig): ProviderAdapter { + return { + name: "fetchonly", + buildRequest: () => ({ url: provider.baseUrl, method: "POST", headers: {}, body: "{}" }), + fetchResponse: async () => new Response("{}", { status: 200 }), + async *parseStream() { + yield { type: "text_delta", text: "search-enabled answer" } as AdapterEvent; + yield { type: "done" } as AdapterEvent; + }, + async parseResponse() { return [{ type: "done" }] as AdapterEvent[]; }, + }; +} mock.module("../../src/server/adapter-resolve", () => ({ ...resolver, resolveAdapter: (provider: OcxProviderConfig, cache?: "none" | "short" | "long") => - provider.adapter === "cursor" ? fixture(provider) : resolveAdapter(provider, cache), + provider.adapter === "cursor" ? fixture(provider) + : provider.adapter === "fetchonly" ? fetchFixture(provider) + : resolveAdapter(provider, cache), })); const pacing = await import("../../src/providers/request-pacing"); const originalWaitForSlot = pacing.waitForProviderRequestSlot; @@ -40,6 +54,20 @@ mock.module("../../src/providers/request-pacing", () => ({ ...pacing, return originalWaitForSlot(...args); }, })); +const sidecarAuth = await import("../../src/server/responses/request-sidecar-auth"); +const prepareResponsesSidecarAuth = sidecarAuth.prepareResponsesSidecarAuth; +let releasedFixtureProbe = false; +mock.module("../../src/server/responses/request-sidecar-auth", () => ({ ...sidecarAuth, + prepareResponsesSidecarAuth: async (...args: Parameters) => { + if (args[0].req.headers.get("x-fixture-probe") !== "held") { + return prepareResponsesSidecarAuth(...args); + } + return { + routedCompaction: false, + openAiSidecar: { releaseProbeLease: () => { releasedFixtureProbe = true; } }, + } as Awaited>; + }, +})); const { handleResponses } = await import("../../src/server/responses"); const originalHome = process.env.OPENCODEX_HOME; let home = ""; @@ -131,6 +159,120 @@ test.each(["image", "video"] as const)("media-only %s bridge still injects its t expect(attempts[0].context.tools?.some(t => t.webSearch)).toBe(false); }); +test("releases a search probe when pre-dispatch validation rejects the request", async () => { + releasedFixtureProbe = false; + const config = { + port: 0, defaultProvider: "cursor", + webSearchSidecar: { backend: "exa", exaApiKey: "fixture-search-key" }, + providers: { + cursor: { adapter: "cursor", baseUrl: "https://api2.cursor.sh", authMode: "oauth", models: ["model"] }, + }, + } as OcxConfig; + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", headers: { "content-type": "application/json", "x-fixture-probe": "held" }, + body: JSON.stringify({ model: "cursor/model", input: [{ + type: "function_call_output", output: "fixture result", + }], tools: [{ type: "web_search" }] }), + }), config, { model: "", provider: "" }); + + expect(response.status).toBe(400); + expect(releasedFixtureProbe).toBe(true); + expect(attempts).toHaveLength(0); +}); + +test("a streamed sidecar response keeps the search probe until the stream settles", async () => { + releasedFixtureProbe = false; + const realFetch = globalThis.fetch; + globalThis.fetch = (async (input: RequestInfo | URL) => { + const url = String(typeof input === "object" && "url" in input ? input.url : input); + if (url.includes("exa")) { + return new Response(JSON.stringify({ results: [{ title: "fixture", url: "https://fixture.test" }] }), + { status: 200, headers: { "content-type": "application/json" } }); + } + return new Response( + 'event: response.completed\ndata: {"type":"response.completed","response":{"output":[]}}\n\n', + { status: 200, headers: { "content-type": "text/event-stream" } }); + }) as typeof fetch; + try { + const config = { + port: 0, defaultProvider: "fetchonly", + webSearchSidecar: { backend: "exa", exaApiKey: "fixture-search-key" }, + providers: { + fetchonly: { adapter: "fetchonly", baseUrl: "https://fetchonly.test/v1", apiKey: "fixture-key", models: ["model"] }, + }, + } as OcxConfig; + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", headers: { "content-type": "application/json", "x-fixture-probe": "held" }, + body: JSON.stringify({ model: "fetchonly/model", input: "search this", stream: true, + tools: [{ type: "web_search" }] }), + }), config, { model: "", provider: "" }); + + expect(response.status).toBe(200); + expect(response.headers.get("content-type")).toContain("event-stream"); + expect(releasedFixtureProbe).toBe(false); + await response.text(); + // The routed model answered without a web_search call, so no sidecar outcome + // settled the lease — the stream's own completion hands the probe back. + expect(releasedFixtureProbe).toBe(true); + } finally { + globalThis.fetch = realFetch; + } +}); + +test("a cancelled streamed sidecar response releases the search probe", async () => { + releasedFixtureProbe = false; + const realFetch = globalThis.fetch; + globalThis.fetch = (async () => new Response( + 'event: response.completed\ndata: {"type":"response.completed","response":{"output":[]}}\n\n', + { status: 200, headers: { "content-type": "text/event-stream" } })) as typeof fetch; + try { + const config = { + port: 0, defaultProvider: "fetchonly", + webSearchSidecar: { backend: "exa", exaApiKey: "fixture-search-key" }, + providers: { + fetchonly: { adapter: "fetchonly", baseUrl: "https://fetchonly.test/v1", apiKey: "fixture-key", models: ["model"] }, + }, + } as OcxConfig; + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", headers: { "content-type": "application/json", "x-fixture-probe": "held" }, + body: JSON.stringify({ model: "fetchonly/model", input: "search this", stream: true, + tools: [{ type: "web_search" }] }), + }), config, { model: "", provider: "" }); + + expect(response.status).toBe(200); + expect(releasedFixtureProbe).toBe(false); + // Client disconnect: the tracked stream's cancel path must settle the lease the same + // way a completed stream does, or the probe stays held until process exit. + await response.body!.cancel(); + expect(releasedFixtureProbe).toBe(true); + } finally { + globalThis.fetch = realFetch; + } +}); + +test("a media-bridge stream releases the search probe when it settles", async () => { + releasedFixtureProbe = false; + events = [[{ type: "text_delta", text: "media answer" }, { type: "done" }]]; + const config = { + port: 0, defaultProvider: "cursor", + images: { bridgeEnabled: true }, + providers: { + cursor: { adapter: "cursor", baseUrl: "https://api2.cursor.sh", authMode: "oauth", models: ["model"] }, + }, + } as OcxConfig; + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", headers: { "content-type": "application/json", "x-fixture-probe": "held" }, + body: JSON.stringify({ model: "cursor/model", input: "draw a fixture", stream: true, + tools: [{ type: "image_generation" }] }), + }), config, { model: "", provider: "" }); + + expect(response.status).toBe(200); + expect(response.headers.get("content-type")).toContain("event-stream"); + expect(releasedFixtureProbe).toBe(false); + await response.text(); + expect(releasedFixtureProbe).toBe(true); +}); + // Streaming only: a first-event 429 replays the turn while the superseded // attempt is still in-flight (the buffered path awaits it before collecting // events, so the race cannot exist there). When that attempt finally returns, From 4497ffe465ddd8ac58b58fb805e196841445af52 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 14:36:06 +0900 Subject: [PATCH 21/75] fix(claude): pin Desktop mode when disabling CLI first-party routing (#6046) Carried from #6046 into merge train round 3. Co-authored-by: Epinephrine --- .../management/agent-settings-routes.ts | 24 +++++++--- structure/config.md | 2 +- structure/gui-and-management-api.md | 2 +- .../claude-management-api.test.ts | 45 +++++++++++++++++++ 4 files changed, 66 insertions(+), 7 deletions(-) diff --git a/src/server/management/agent-settings-routes.ts b/src/server/management/agent-settings-routes.ts index 16c60286b7d..65f9859672b 100644 --- a/src/server/management/agent-settings-routes.ts +++ b/src/server/management/agent-settings-routes.ts @@ -1558,7 +1558,8 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise catch { return jsonResponse({ error: "Claude settings are unreadable", code: "unreadable" }, 500); } type FirstPartyMutation = | { refusal: { error: string; code: "intercept_disabled" | "intercept_unavailable" } } - | { claudeCode: OcxConfig["claudeCode"]; previous: { present: boolean; value: boolean }; pinnedMode: "first-party" | "gateway" | undefined }; + | { claudeCode: OcxConfig["claudeCode"]; previous: { present: boolean; value: boolean }; + pinnedMode: "first-party" | "gateway" | undefined; retainedAmbiguous: boolean }; let outcome: ReturnType>; try { outcome = mutatePersistedConfig(persisted => { @@ -1573,14 +1574,23 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise const before = structuredClone(persisted); const previous = { present: Object.hasOwn(persisted.claudeCode ?? {}, "cliFirstParty"), value: persisted.claudeCode?.cliFirstParty === true }; - const pinnedMode = body.cliFirstParty && persisted.claudeCode?.desktopMode === undefined - ? resolveClaudeDesktopMode(before, observeClaudeDesktopMode(before)) : undefined; + // Pin the mode Desktop resolves to *after* this mutation: while cliFirstParty is set the + // shared env is suppressed as Desktop evidence, so an opt-out observed with the flag still + // on would pin gateway and disconnect a Desktop install that predates the marker. + const observedBefore = structuredClone(before); + if (!body.cliFirstParty) delete observedBefore.claudeCode?.cliFirstParty; + const pinnedMode = persisted.claudeCode?.desktopMode === undefined + ? resolveClaudeDesktopMode(before, observeClaudeDesktopMode(observedBefore)) : undefined; const nextBlock = { ...(persisted.claudeCode ?? {}) }; if (body.cliFirstParty) nextBlock.cliFirstParty = true; else delete nextBlock.cliFirstParty; if (pinnedMode) nextBlock.desktopMode = pinnedMode; commitClaudeCodeBlock(persisted, nextBlock); - return { changed: true, value: { claudeCode: structuredClone(persisted.claudeCode), previous, pinnedMode } }; + // An opt-out that pins first-party from the shared env cannot tell whether that env was + // Desktop's or a hand-configured CLI-only one; the env is retained (Desktop keeps its + // route) and the caller is warned so it can pin gateway explicitly to release it. + const retainedAmbiguous = !body.cliFirstParty && previous.value && pinnedMode === "first-party"; + return { changed: true, value: { claudeCode: structuredClone(persisted.claudeCode), previous, pinnedMode, retainedAmbiguous } }; }); } catch { return jsonResponse({ error: "Could not save Claude settings", code: "write_failed" }, 500); } if (outcome.status === "unavailable") return jsonResponse({ error: "Could not save Claude settings", code: "write_failed" }, 500); @@ -1630,7 +1640,11 @@ export async function handleAgentSettingsRoutes(ctx: ManagementContext): Promise const residual = !finalDesired.desktop && !finalDesired.cli && readFirstPartyProxyStatus(config, bound?.proxyPort ?? null) !== "none"; return jsonResponse({ ok: true, enabled: config.claudeCode?.enabled !== false, - cliFirstParty: body.cliFirstParty, warnings: residual ? ["settings_residual"] : [] }); + cliFirstParty: body.cliFirstParty, + warnings: [ + ...(committed.retainedAmbiguous ? ["shared_proxy_retained"] : []), + ...(residual ? ["settings_residual"] : []), + ] }); } for (const field of ["webSearchSidecar", "visionSidecar"] as const) { const section = body[field]; diff --git a/structure/config.md b/structure/config.md index 98490672149..f6acb718130 100644 --- a/structure/config.md +++ b/structure/config.md @@ -125,7 +125,7 @@ record; it does not call `loadConfig`, mutate permissions, or import the write-c All config publication continues through the existing required ACL-hardened writers above. `claudeCode.desktopProfile` follows the same preserve-the-rest rule. JSON `null` (or any non-string) `appliedFingerprint` / `appliedAt` is treated as unset. A profile that is still invalid after that is dropped as a whole — `src/config/salvage.ts` already does this for independent `routingProfiles` / `combos` entries — so one bad Desktop marker cannot replace the operator's providers with `getDefaultConfig()`. A `claudeCode` value that is not an object still fails the document, because there is no safe subtree to keep. -`claudeCode.cliFirstParty` is an optional boolean in `src/types/config.ts`. The schema passes it through; the load normalizer (`src/config/load-degrade.ts`) drops a non-boolean hand edit, every reader treats only `true` as on, and `PUT /api/claude-code` accepts only a boolean. Absence means off. It is independent of `claudeCode.desktopMode`; enabling CLI first-party pins an absent Desktop mode from a pre-write observation, before writing the shared settings env, so later Desktop inference cannot mistake a CLI-only env for Desktop intent. The flag is written only by a standalone `PUT /api/claude-code { cliFirstParty }`, including `ocx claude config set --first-party`; enabling it pins an absent `desktopMode` in the same persisted mutation. The shared settings proxy status follows the ordered classifier in `src/claude/first-party-settings.ts`: unreadable settings are `unknown`; absent or unrecognized proxy URLs are `none`; a token-bearing opencodex URL beside a foreign CA is `foreign`, while a tokenless loopback URL beside that CA is `local` with unconfirmed ownership. An attributed proxy with no bound listener is `stopped`; a usable applied pair on a bound listener is `disabled` when Claude routing is ineligible and `live` when eligible; remaining mismatches are `broken` regardless of eligibility. Inspection never mints a token. A separate `ocx ensure` may write a config-derived port while this server remains bound elsewhere; status is then `broken` until the server restarts or ensure runs after restart. +`claudeCode.cliFirstParty` is an optional boolean in `src/types/config.ts`. The schema passes it through; the load normalizer (`src/config/load-degrade.ts`) drops a non-boolean hand edit, every reader treats only `true` as on, and `PUT /api/claude-code` accepts only a boolean. Absence means off. It is independent of `claudeCode.desktopMode`; changing CLI first-party pins an absent Desktop mode from the observation that will apply after the flag flips — an opt-out observes with `cliFirstParty` already cleared, so a shared env that predates the marker stays attributed to Desktop instead of being pinned `gateway` and removed from under it, and an owned env cannot be mistaken for CLI-only intent. The flag is written by a standalone `PUT /api/claude-code { cliFirstParty }`, including `ocx claude config set --first-party`; the mutation pins an absent `desktopMode` at the same time. The shared settings proxy status follows the ordered classifier in `src/claude/first-party-settings.ts`: unreadable settings are `unknown`; absent or unrecognized proxy URLs are `none`; a token-bearing opencodex URL beside a foreign CA is `foreign`, while a tokenless loopback URL beside that CA is `local` with unconfirmed ownership. An attributed proxy with no bound listener is `stopped`; a usable applied pair on a bound listener is `disabled` when Claude routing is ineligible and `live` when eligible; remaining mismatches are `broken` regardless of eligibility. Inspection never mints a token. A separate `ocx ensure` may write a config-derived port while this server remains bound elsewhere; status is then `broken` until the server restarts or ensure runs after restart. The former `showCodexSparkQuota` key is inert passthrough data when loading an old config. It is absent from the typed settings contract and cannot re-enable Spark quota through the management API. Retirement does not migrate user-selected model ids or erase usage history. diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index ef81291fd07..28c65f4e7a9 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -229,7 +229,7 @@ per-request first-party callback reads that live object; a failed write leaves i | Effort and fallback | `src/server/management/agent-settings-routes.ts` — `GET/PUT /api/effort-caps`, `/api/subagent-models`, `/api/subagent-model-fallback`. Caps clamp; they do not reject. | | Grok and Claude integrations | `src/server/management/agent-settings-routes.ts` — `GET /api/grok`, `PUT /api/grok/selection`, `POST /api/grok/apply`, `GET/PUT /api/claude-desktop`, `POST /api/claude-desktop/apply` (`mode`: `gateway` default for new installs, `first-party` opt-in, or legacy shapes), `GET /api/claude-desktop/status` (`mode`, `riskWarning`, `firstParty`), `GET/PUT /api/claude-code`. Gateway apply writes an external app's profile, so its status probe must read the same resolved path it writes (see [`responses.md`](transports/responses.md)); first-party apply writes only the Claude Code proxy env, see [`clients/claude-desktop.md`](clients/claude-desktop.md#desktop-modes-gateway-and-first-party). `gui/src/pages/ClaudeDesktop.tsx` renders the mode selector and sends the chosen `mode` with apply. `PUT /api/claude-code` re-runs macOS system-env reconciliation (`src/server/system-env.ts`) whenever the body carries `systemEnv`, `authMode`, a model slot or a lever field: keys opencodex tracks as injected are refreshed or unset once the config stops producing them, and a launchd value the user set before injection is never touched. | -`GET /api/claude-code` reports `cliFirstParty`, `desktopFirstParty`, `cliFirstPartyApplied`, `interceptEligible`, `interceptRunning`, and the eight-value `sharedProxy: FirstPartyProxyStatus` from observed settings and the bound listener. `interceptEligible = claudeInterceptEnabled(config)` uses the same GET snapshot as `sharedProxy` and `interceptRunning`; the latter remains bound listener present AND eligible. The ordered classifier gives unreadable → `unknown`, absent or non-loopback URL → `none`, foreign CA with an opencodex token → `foreign`, foreign CA with a tokenless loopback URL → `local`, no bound listener → `stopped`, ineligible applied settings at the bound port → `disabled`, other ineligible or stale/mismatched settings → `broken`, and eligible applied settings at the bound port → `live`. `cliFirstPartyApplied` requires CLI intent and `live`. `PUT /api/claude-code` accepts standalone `cliFirstParty`; CLI-on repeats eligibility and port checks inside the locked persisted mutation, reconciles, and conditionally rolls back its own fields on failure. CLI-off deletes intent but retains an env Desktop still desires. A successful nothing-desired reconcile returns `settings_residual` for every status except `none`, including `local`; unreadable cleanup returns the coded 500. `enabled:false` alone leaves the env untouched, and a mixed body returns 400 before save. GUI normalization maps only missing `sharedProxy:undefined` to `none`, invalid statuses including `null` to `unknown`; `interceptEligible:undefined` from an older cache maps to `true`, while present values use `=== true`. The source coverage map checks all eight statuses. Notice order is unknown, foreign, local, residual when undesired, disabled, routingOff for stopped/broken with ineligible routing, stopped, broken, notApplied, shared, null. Unknown copy states uncertainty, local copy names the unconfirmed 127.0.0.1 proxy and manual HTTPS_PROXY removal, disabled copy retains the first-party-off remedy, foreign copy directs manual CA/proxy repair, routingOff says to restore Claude routing or turn first-party off, stopped says to start opencodex, and eligible broken advises `ocx ensure` or restart. +`GET /api/claude-code` reports `cliFirstParty`, `desktopFirstParty`, `cliFirstPartyApplied`, `interceptEligible`, `interceptRunning`, and the eight-value `sharedProxy: FirstPartyProxyStatus` from observed settings and the bound listener. `interceptEligible = claudeInterceptEnabled(config)` uses the same GET snapshot as `sharedProxy` and `interceptRunning`; the latter remains bound listener present AND eligible. The ordered classifier gives unreadable → `unknown`, absent or non-loopback URL → `none`, foreign CA with an opencodex token → `foreign`, foreign CA with a tokenless loopback URL → `local`, no bound listener → `stopped`, ineligible applied settings at the bound port → `disabled`, other ineligible or stale/mismatched settings → `broken`, and eligible applied settings at the bound port → `live`. `cliFirstPartyApplied` requires CLI intent and `live`. `PUT /api/claude-code` accepts standalone `cliFirstParty`; CLI-on repeats eligibility and port checks inside the locked persisted mutation, reconciles, and conditionally rolls back its own fields on failure. CLI-off deletes intent, pins an absent `desktopMode` from the post-clear observation, and retains an env Desktop still desires; when that retained env's ownership was ambiguous (an owned shared proxy suppressed by the flag), the success response warns `shared_proxy_retained` so the operator can pin `gateway` explicitly to release it. A successful nothing-desired reconcile returns `settings_residual` for every status except `none`, including `local`; unreadable cleanup returns the coded 500. `enabled:false` alone leaves the env untouched, and a mixed body returns 400 before save. GUI normalization maps only missing `sharedProxy:undefined` to `none`, invalid statuses including `null` to `unknown`; `interceptEligible:undefined` from an older cache maps to `true`, while present values use `=== true`. The source coverage map checks all eight statuses. Notice order is unknown, foreign, local, residual when undesired, disabled, routingOff for stopped/broken with ineligible routing, stopped, broken, notApplied, shared, null. Unknown copy states uncertainty, local copy names the unconfirmed 127.0.0.1 proxy and manual HTTPS_PROXY removal, disabled copy retains the first-party-off remedy, foreign copy directs manual CA/proxy repair, routingOff says to restore Claude routing or turn first-party off, stopped says to start opencodex, and eligible broken advises `ocx ensure` or restart. | File-integration plans | `src/server/management/integration-routes.ts` and `aside-profile-routes.ts` — `POST /api/client-integrations/preview`, `POST /api/client-integrations/restore/preview`, and `POST /api/client-integrations/aside/profiles/{profileId}/preview`. Management-authenticated, declared non-mutating, and they write nothing: no snapshot, no lock, no maintenance, no recovery. They answer `409 integration_preview_unavailable` rather than gathering a model roster, because discovery refreshes credentials and writes the provider cache. Responses carry only declared managed schema paths, closed change kinds and an opaque fingerprint; no value, filesystem location or selected member identity appears. Mutation routes accept `operation` and `planFingerprint` together or not at all, reject a half-bound request and an operation that disagrees with the change, and answer `409 integration_preview_stale` with a freshly computed plan. Binding is an optimistic token, never authorization. [The integration contract](clients/integrations.md) owns the ordering. | | Grok reset coupons | `src/server/management/grok-coupon-routes.ts` — `GET /api/grok/reset-coupons`, `POST /api/grok/reset-coupons/consume`. The dashboard owner is `gui/src/hooks/useGrokResetCoupons.ts` with `gui/src/components/provider-workspace/GrokResetCoupons.tsx`, wired into the xAI OAuth rows of `ProviderAuthPanel`. Redemption truth is the settled ledger `code`, not the HTTP status: a replayed failure returns 200 with `replayed: true`. See [`providers/xai-grok.md`](providers/xai-grok.md). | diff --git a/tests/claude-integration/claude-management-api.test.ts b/tests/claude-integration/claude-management-api.test.ts index 679b3a7e83e..341f703501d 100644 --- a/tests/claude-integration/claude-management-api.test.ts +++ b/tests/claude-integration/claude-management-api.test.ts @@ -367,6 +367,51 @@ test("CLI-off reports a tokenless local proxy with foreign CA as residue", async } finally { await server.stop(true); } }); +test("CLI-off keeps a shared env Desktop could own, pinning first-party instead of removing it", async () => { + // A legacy install can carry an owned shared proxy and cliFirstParty without a desktopMode + // marker. The env is ambiguous while the flag is set, so opt-out must not pin gateway and + // remove a connection Desktop may still be using. + const current = loadConfig(); + current.port = 10100; + current.claudeCode = { ...current.claudeCode, cliFirstParty: true }; + saveConfig(current); + const settingsPath = join(process.env.CLAUDE_CONFIG_DIR!, "settings.json"); + mkdirSync(process.env.CLAUDE_CONFIG_DIR!, { recursive: true }); + writeFileSync(settingsPath, JSON.stringify({ env: desktopFirstPartyTarget(current).env })); + const server = startServer(0); + try { + const response = await fetch(new URL("/api/claude-code", server.url), { + method: "PUT", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ cliFirstParty: false }), + }); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({ cliFirstParty: false, warnings: ["shared_proxy_retained"] }); + const claudeCode = loadConfig().claudeCode; + expect(claudeCode?.cliFirstParty).toBeUndefined(); + expect(claudeCode?.desktopMode).toBe("first-party"); + expect(JSON.parse(readFileSync(settingsPath, "utf8")).env).toBeDefined(); + expect(await (await fetch(new URL("/api/claude-code", server.url))).json()) + .toMatchObject({ cliFirstParty: false, desktopFirstParty: true }); + } finally { await server.stop(true); } +}); + +test("CLI-off pins gateway and clears the env when nothing on disk is ours", async () => { + const current = loadConfig(); + current.port = 10100; + current.claudeCode = { ...current.claudeCode, cliFirstParty: true }; + saveConfig(current); + const server = startServer(0); + try { + const response = await fetch(new URL("/api/claude-code", server.url), { + method: "PUT", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ cliFirstParty: false }), + }); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({ cliFirstParty: false, warnings: [] }); + expect(loadConfig().claudeCode).toMatchObject({ desktopMode: "gateway" }); + expect(await (await fetch(new URL("/api/claude-code", server.url))).json()) + .toMatchObject({ cliFirstParty: false, desktopFirstParty: false, sharedProxy: "none" }); + } finally { await server.stop(true); } +}); + test("CLI-off persists intent despite unreadable settings cleanup", async () => { const current = loadConfig(); current.claudeCode = { ...current.claudeCode, cliFirstParty: true, desktopMode: "gateway" }; From a3d40e1f63f22c55c38553d652e08cd1b9652ee8 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:36:58 +0900 Subject: [PATCH 22/75] fix(cli): show when Claude Desktop keeps the shared proxy env after first-party off Follow-up to #6046: the management route reports shared_proxy_retained, but the human CLI output dropped it. --- src/cli/integrations.ts | 9 ++++++++- tests/cli/claude-config-first-party.test.ts | 17 +++++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/src/cli/integrations.ts b/src/cli/integrations.ts index 9afcae078a3..1caa2be0cb0 100644 --- a/src/cli/integrations.ts +++ b/src/cli/integrations.ts @@ -126,7 +126,14 @@ export async function handleClaudeConfigCommand(argv: string[], deps: RuntimeApi } if (Object.keys(body).length === 0) throw new CliUsageError("at least one Claude setting is required", CLAUDE_USAGE); const result = await runtimeRequest("/api/claude-code", { method: "PUT", body: JSON.stringify(body) }, deps); - printData(result, wantsJson, ["Claude Code settings updated."]); + const warnings = (result as { warnings?: unknown }).warnings; + const retained = Array.isArray(warnings) && warnings.includes("shared_proxy_retained"); + printData(result, wantsJson, [ + "Claude Code settings updated.", + // The route kept the shared proxy env because Claude Desktop may still rely on it. Say so, + // or an operator who turned first-party off believes the local interception is gone. + ...(retained ? ["Warning: Claude Desktop still uses the shared proxy settings. Run `ocx claude desktop apply --gateway` to release them."] : []), + ]); }); } diff --git a/tests/cli/claude-config-first-party.test.ts b/tests/cli/claude-config-first-party.test.ts index 246571563f3..066408fa5ec 100644 --- a/tests/cli/claude-config-first-party.test.ts +++ b/tests/cli/claude-config-first-party.test.ts @@ -16,6 +16,23 @@ for (const [value, expected] of [["on", true], ["off", false]] as const) { }); } +test("--first-party off tells the operator when Claude Desktop keeps the shared proxy env", async () => { + const lines: string[] = []; + const original = console.log; + console.log = (...args: unknown[]) => { lines.push(args.join(" ")); }; + try { + const code = await handleClaudeConfigCommand(["set", "--first-party", "off"], { + baseUrl: "http://127.0.0.1:1", + fetchImpl: async () => Response.json({ ok: true, warnings: ["shared_proxy_retained"] }), + }); + expect(code).toBe(0); + } finally { + console.log = original; + } + expect(lines[0]).toBe("Claude Code settings updated."); + expect(lines[1]).toContain("ocx claude desktop apply --gateway"); +}); + for (const args of [ ["set", "--first-party", "maybe"], ["set", "--first-party", "on", "--system-env", "on"], From 79faa24ee27e324d92a72b2545ce70a4e40ef5f9 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 14:37:01 +0900 Subject: [PATCH 23/75] fix(service): match canonical Codex home ownership (#6036) Carried from #6036 into merge train round 3. Co-authored-by: Epinephrine --- .../content/docs/reference/cli/lifecycle.md | 6 ++ .../native/ownership-preflight.ts | 58 +++++++++++++++---- src/service.ts | 4 +- src/service/guards.ts | 9 ++- src/service/state.ts | 31 +++++++++- structure/codex-home.md | 7 ++- structure/runtime.md | 2 +- .../codex-integration/codex-home-wsl.test.ts | 54 ++++++++++++++++- ...ex-service-manager-probe-hardening.test.ts | 4 ++ .../codex-service-manager-probe.test.ts | 45 +++++++++++++- tests/service/service-sqlite-home.test.ts | 22 +++++++ .../service-wsl-home-ownership.test.ts | 1 + 12 files changed, 216 insertions(+), 27 deletions(-) diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index 413c93e0264..8f66ede7b01 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -490,6 +490,12 @@ supersedes it rather than replacing it. A state file with no ownership record means the CLI installation owns the runtime, which is what every installation made before this feature is in. Nothing changes for you until an app takes over. +Home paths inside a state record are compared with the current home by the physical directory they +resolve to, not just their spelling. A junction or symlink recorded under an older install still +names the same home and keeps working after the move; an alias that no longer resolves is only +treated as a different home when its recorded spelling also differs from the current one, so a +stale mount still produces the foreign-owner refusal instead of silently claiming the runtime. + While something other than this CLI owns the runtime, the subcommands that would **activate** your registration refuse instead: diff --git a/src/integrations/native/ownership-preflight.ts b/src/integrations/native/ownership-preflight.ts index 7ca9ca9771f..065870127d2 100644 --- a/src/integrations/native/ownership-preflight.ts +++ b/src/integrations/native/ownership-preflight.ts @@ -18,9 +18,11 @@ import { import { currentServiceHomes, inspectServiceStateEvidence, - serviceHomeMatches, + compareServicePathToInstall, type ServiceStateEvidence, + type ServicePathComparison, } from "../../service"; +import type { CodexHomeDeps } from "../../codex/home"; import { createWindowsTaskListingCache, inspectServiceManagerInstallation, @@ -70,15 +72,25 @@ export interface OwnershipInspection { readonly reason: string; } -function claimNamesDifferentHome( +function claimComparesToCurrentHomes( claim: ServiceManagerClaim, current: { codexHome: string; opencodexHome: string }, -): boolean { + deps: CodexHomeDeps, +): ServicePathComparison { // A definition that OMITS a home is not a definition that disagrees about it: // an install run without CODEX_HOME set writes no such key at all. - if (claim.homes.codexHome !== null && !serviceHomeMatches(claim.homes.codexHome, current.codexHome)) return true; - if (claim.homes.opencodexHome !== null && !serviceHomeMatches(claim.homes.opencodexHome, current.opencodexHome)) return true; - return false; + let indeterminate = false; + if (claim.homes.codexHome !== null) { + const verdict = compareServicePathToInstall(claim.homes.codexHome, current.codexHome, deps); + if (verdict === "different") return "different"; + indeterminate ||= verdict === "unknown"; + } + if (claim.homes.opencodexHome !== null) { + const verdict = compareServicePathToInstall(claim.homes.opencodexHome, current.opencodexHome, deps); + if (verdict === "different") return "different"; + indeterminate ||= verdict === "unknown"; + } + return indeterminate ? "unknown" : "same"; } /** @@ -109,6 +121,8 @@ export interface OwnershipDeps extends ProbeDeps { */ readonly statePaths?: readonly string[]; readonly currentHomes?: { codexHome: string; opencodexHome: string }; + /** Test seam for resolving recorded home aliases to their physical directory. */ + readonly realpathSync?: (path: string) => string; } export function inspectNativeCodexOwnership(deps: OwnershipDeps = {}): OwnershipInspection { @@ -130,22 +144,35 @@ export function inspectNativeCodexOwnership(deps: OwnershipDeps = {}): Ownership // Mirrors that disagree with each other are not a majority vote. for (const one of valid) { for (const other of valid) { - if (!serviceHomeMatches(one.state.codexHome, other.state.codexHome) - || !serviceHomeMatches(one.state.opencodexHome, other.state.opencodexHome)) { + if (compareServicePathToInstall(one.state.codexHome, other.state.codexHome, deps) !== "same" + || compareServicePathToInstall(one.state.opencodexHome, other.state.opencodexHome, deps) !== "same") { return { ownership: "unknown", reason: "two service state files disagree about which homes are installed" }; } } } - const foreign = valid.find(e => - !serviceHomeMatches(e.state.codexHome, current.codexHome) - || !serviceHomeMatches(e.state.opencodexHome, current.opencodexHome)); + const evidenceComparesDifferent = (e: Extract): boolean => + compareServicePathToInstall(e.state.codexHome, current.codexHome, deps) === "different" + || compareServicePathToInstall(e.state.opencodexHome, current.opencodexHome, deps) === "different"; + const evidenceComparesIndeterminate = (e: Extract): boolean => + compareServicePathToInstall(e.state.codexHome, current.codexHome, deps) === "unknown" + || compareServicePathToInstall(e.state.opencodexHome, current.opencodexHome, deps) === "unknown"; + const foreign = valid.find(evidenceComparesDifferent); if (foreign) { return { ownership: "foreign", reason: `a service is installed for CODEX_HOME=${foreign.state.codexHome} / OPENCODEX_HOME=${foreign.state.opencodexHome}`, }; } + // A resolution that could not run (EACCES, EPERM, a vanished directory, transient I/O) + // proves neither same nor different — report it as unknown, never as foreign. + const indeterminateEvidence = valid.find(evidenceComparesIndeterminate); + if (indeterminateEvidence) { + return { + ownership: "unknown", + reason: `a recorded home in ${indeterminateEvidence.path} could not be resolved for comparison`, + }; + } // The manager assets live under the effective OPENCODEX_HOME. Production // callers do not inject ProbeDeps.configDir, so derive it from the same @@ -162,7 +189,7 @@ export function inspectNativeCodexOwnership(deps: OwnershipDeps = {}): Ownership return { ownership: "unknown", reason: "more than one service manager holds a registration for this proxy" }; } if (manager.kind === "present") { - const disagreeing = manager.claims.find(claim => claimNamesDifferentHome(claim, current)); + const disagreeing = manager.claims.find(claim => claimComparesToCurrentHomes(claim, current, deps) === "different"); if (disagreeing) { /* * The state file says this home and the definition says another. An @@ -175,6 +202,13 @@ export function inspectNativeCodexOwnership(deps: OwnershipDeps = {}): Ownership reason: `${disagreeing.backend} is installed from ${disagreeing.definitionPath}, which names different homes than the recorded service state`, }; } + const indeterminateClaim = manager.claims.find(claim => claimComparesToCurrentHomes(claim, current, deps) === "unknown"); + if (indeterminateClaim) { + return { + ownership: "unknown", + reason: `the homes recorded in ${indeterminateClaim.definitionPath} could not be resolved for comparison`, + }; + } // A manager backend that disagrees with the recorded state (e.g. state says // native/WinSW but a scheduler task is found) is an interrupted backend // switch: it does not prove which manager owns the installation. v1 state diff --git a/src/service.ts b/src/service.ts index 8be7805d0ad..3706b1d3fe9 100644 --- a/src/service.ts +++ b/src/service.ts @@ -6,8 +6,8 @@ * restore it via the command. */ -export type { ServiceBackend, ServiceInstallState, ServiceStateEvidence, ServiceStateResolution, ServiceOwner, ServiceOwnership, ServiceOwnershipSubject, ServiceOwnershipResolution, ServiceStateSwapDeps, RecordServiceOwnerRequest, RecordServiceOwnerDeps, ReleaseServiceOwnerDeps, RemoveServiceStateDeps } from "./service/state"; -export { SERVICE_MANAGED_ENV, SERVICE_OWNERSHIP_PROTOCOL_VERSION, SERVICE_OWNERSHIP_MINIMUM_CLI_VERSION, stableLauncherEntry, serviceLogPath, serviceStatePaths, serviceStatePathsForOpenCodexHome, parseServiceInstallState, parseServiceOwnership, inspectServiceStateEvidence, resolveServiceState, currentServiceHomes, serviceHomeMatches, serviceCodexHomeMatchesInstall, readServiceBackend, serviceReinstallArgs, serviceInstallArgs, ServiceStateConflictError, ServiceOwnershipSubjectMismatchError, ServiceOwnershipSubjectUnknownError, ServiceTakeoverCompatibilityChangedError, swapServiceInstallState, removeServiceInstallStateRecords, serviceOwnership, resolveServiceOwnership, sameServiceOwnershipSubject, desktopOwnsService, ownershipGrantedTo, recordServiceOwner, releaseServiceOwner } from "./service/state"; +export type { ServiceBackend, ServiceInstallState, ServiceStateEvidence, ServiceStateResolution, ServiceOwner, ServiceOwnership, ServiceOwnershipSubject, ServiceOwnershipResolution, ServicePathComparison, ServiceStateSwapDeps, RecordServiceOwnerRequest, RecordServiceOwnerDeps, ReleaseServiceOwnerDeps, RemoveServiceStateDeps } from "./service/state"; +export { SERVICE_MANAGED_ENV, SERVICE_OWNERSHIP_PROTOCOL_VERSION, SERVICE_OWNERSHIP_MINIMUM_CLI_VERSION, stableLauncherEntry, serviceLogPath, serviceStatePaths, serviceStatePathsForOpenCodexHome, parseServiceInstallState, parseServiceOwnership, inspectServiceStateEvidence, resolveServiceState, currentServiceHomes, serviceHomeMatches, serviceCodexHomeMatchesInstall, servicePathMatchesInstall, compareServicePathToInstall, readServiceBackend, serviceReinstallArgs, serviceInstallArgs, ServiceStateConflictError, ServiceOwnershipSubjectMismatchError, ServiceOwnershipSubjectUnknownError, ServiceTakeoverCompatibilityChangedError, swapServiceInstallState, removeServiceInstallStateRecords, serviceOwnership, resolveServiceOwnership, sameServiceOwnershipSubject, desktopOwnsService, ownershipGrantedTo, recordServiceOwner, releaseServiceOwner } from "./service/state"; export type { OwnershipMutationLeaseOptions, OwnershipMutationLease } from "./service/ownership-mutation-lease.mjs"; export { acquireOwnershipMutationLease, withOwnershipMutationLease } from "./service/ownership-mutation-lease.mjs"; export type { ManagingCliRole, ManagingCliObservation, RegisteredManagingCliInvocation, ServiceTakeoverCompatibilityInput, ServiceTakeoverCompatibility } from "./service/ownership-compatibility"; diff --git a/src/service/guards.ts b/src/service/guards.ts index 436d053953f..93ad78e174f 100644 --- a/src/service/guards.ts +++ b/src/service/guards.ts @@ -10,7 +10,7 @@ import { recordOwnedConfigPath } from "../lib/config-ownership"; import { isTestHomeGuardArmed } from "../lib/test-home-guard"; import { diagnoseService } from "./diagnostics"; import type { ServiceDiagnostic } from "./diagnostics"; -import { currentCodexHome, currentOpenCodexHome, normalizePathForCompare, resolveServiceState, serviceCodexHomeMatchesInstall } from "./state"; +import { currentCodexHome, currentOpenCodexHome, resolveServiceState, serviceCodexHomeMatchesInstall, servicePathMatchesInstall } from "./state"; import { resolveCodexSqliteHome } from "../codex/paths"; import type { CodexHomeDeps } from "../codex/home"; import { isLoopbackHostname } from "../codex/loopback-target"; @@ -58,9 +58,8 @@ export function assertServiceEnvironmentMatchesInstall(deps: CodexHomeDeps = {}) `Rerun with CODEX_HOME=${state.codexHome} so native Codex restore updates the recorded home.`, ); } - const expectedOpenCodexHome = normalizePathForCompare(state.opencodexHome); - const actualOpenCodexHome = normalizePathForCompare(currentOpenCodexHome()); - if (expectedOpenCodexHome !== actualOpenCodexHome) { + const actualOpenCodexHome = currentOpenCodexHome(); + if (!servicePathMatchesInstall(state.opencodexHome, actualOpenCodexHome, deps)) { throw new ServiceOwnershipError( `Service was installed with OPENCODEX_HOME=${state.opencodexHome}, but current OPENCODEX_HOME=${currentOpenCodexHome()}. ` + "Run the service command from the same OpenCodex home so service state and secrets match.", @@ -68,7 +67,7 @@ export function assertServiceEnvironmentMatchesInstall(deps: CodexHomeDeps = {}) } if (state.codexSqliteHome !== undefined) { const actualCodexSqliteHome = resolveCodexSqliteHome({ codexHome: actualCodexHome }); - if (normalizePathForCompare(state.codexSqliteHome) !== normalizePathForCompare(actualCodexSqliteHome)) { + if (!servicePathMatchesInstall(state.codexSqliteHome, actualCodexSqliteHome, deps)) { throw new ServiceOwnershipError( `Service was installed with Codex SQLite home=${state.codexSqliteHome}, but the current Codex SQLite home=${actualCodexSqliteHome}. ` + "Run the service command with the same sqlite_home configuration and CODEX_SQLITE_HOME so native Codex history restore updates the correct database.", diff --git a/src/service/state.ts b/src/service/state.ts index f3844feacea..d4f390773e4 100644 --- a/src/service/state.ts +++ b/src/service/state.ts @@ -1,4 +1,4 @@ -import { accessSync, constants as fsConstants, existsSync, readFileSync, statSync, unlinkSync, writeFileSync } from "node:fs"; +import { accessSync, constants as fsConstants, existsSync, readFileSync, realpathSync, statSync, unlinkSync, writeFileSync } from "node:fs"; import { homedir } from "node:os"; import { delimiter, dirname, isAbsolute, join, posix, resolve, win32 } from "node:path"; import { expandUserPath, getConfigDir } from "../config"; @@ -857,8 +857,35 @@ export function serviceHomeMatches(a: string, b: string): boolean { return normalizePathForCompare(a) === normalizePathForCompare(b); } +export type ServicePathComparison = "same" | "different" | "unknown"; + +/** + * Tri-state physical-home compare. A realpath failure (EACCES, EPERM, a + * vanished directory, transient I/O) is "unknown", not "different": callers + * deciding whether a home is foreign must not turn an unreadable resolution + * into a definitive mismatch. Lifecycle guards may still fail closed on + * "unknown". + */ +export function compareServicePathToInstall(recorded: string, current: string, deps: CodexHomeDeps = {}): ServicePathComparison { + if (serviceHomeMatches(recorded, current)) return "same"; + const realpath = deps.realpathSync ?? realpathSync; + try { + return serviceHomeMatches(realpath(recorded), realpath(current)) ? "same" : "different"; + } catch { + return "unknown"; + } +} + +/** Lexical compare first; when spellings differ, compare the directories both resolve to so a + * junction or symlink spelling recorded by an older install still names the same home. + * Fails closed on an indeterminate resolution — ownership classification needs the + * tri-state {@link compareServicePathToInstall} instead. */ +export function servicePathMatchesInstall(recorded: string, current: string, deps: CodexHomeDeps = {}): boolean { + return compareServicePathToInstall(recorded, current, deps) === "same"; +} + export function serviceCodexHomeMatchesInstall(recordedHome: string, deps: CodexHomeDeps = {}): boolean { - return serviceHomeMatches(recordedHome, currentCodexHome(deps)); + return servicePathMatchesInstall(recordedHome, currentCodexHome(deps), deps); } /** Single accessor for backend-sensitive service code — v1/legacy state maps to scheduler. */ diff --git a/structure/codex-home.md b/structure/codex-home.md index 74c4e980a20..d14ee6bfcdc 100644 --- a/structure/codex-home.md +++ b/structure/codex-home.md @@ -95,7 +95,12 @@ to the single discoverable Windows Desktop home; recording Linux `~/.codex` inst later repair or uninstall look foreign even though the service and runtime were started from the same environment. A record written before that discovery still names Linux `~/.codex`; service commands refuse it and name the recorded home to rerun with, because stop and repair would otherwise -restore a different home. An explicit `CODEX_HOME` remains authoritative; nothing migrates implicitly. +restore a different home. A recorded spelling that still resolves to the same physical directory — +a junction or symlink alias — counts as the same home for the ownership check, the recorded SQLite +home, and the unattended ownership preflight. Access or transient I/O errors are unknown rather +than foreign in preflight; distinct spellings with missing/non-directory paths remain mismatches. +Lifecycle guards still fail closed on unknown and can throw `ServiceOwnershipError`. +An explicit `CODEX_HOME` remains authoritative; nothing migrates implicitly. > Decision record: [ADR-0006](decisions/ADR-0006-codex-home.md) diff --git a/structure/runtime.md b/structure/runtime.md index 693f9926b41..0d26fe16cf3 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -140,7 +140,7 @@ does not perform OAuth, and runtime credential resolution rereads the owned sour | `src/types.ts` | Shared config, parsed request, adapter, and event types. | | `src/reasoning-effort.ts` | Codex reasoning-level definitions (`low`/`medium`/`high`/`xhigh`), per-model effort mapping, and catalog effort sanitization. | | `src/codex/shim.ts` | Codex autostart shim: replaces the `codex` binary with a wrapper that auto-starts the proxy on demand. It skips startup for management subcommands even when value-taking global flags precede the subcommand, and transactionally restores complete, stable external launcher replacements without a watcher or PATH rediscovery. | -| `src/service.ts` | OS service manager (macOS launchd, Linux systemd, Windows schtasks): always-on proxy with crash restart. Facade over the `src/service/` leaves — `src/service/launchd.ts`, `src/service/systemd.ts`, `src/service/windows-ops.ts`, `src/service/windows-scheduler.ts`, `src/service/windows-taskxml.ts`, `src/service/state.ts`, `src/service/guards.ts`, `src/service/health.ts`, `src/service/repair.ts`, `src/service/orchestration.ts`, `src/service/diagnostics.ts`, `src/service/cli.ts`. Elevated Task Scheduler repair stages bounded payloads; the unelevated launcher pins every namespace ancestor and payload with non-reparse handles that deny write/delete sharing on the payload and delete sharing on each ancestor until UAC processing exits. | +| `src/service.ts` | OS service manager (macOS launchd, Linux systemd, Windows schtasks): always-on proxy with crash restart. Facade over the `src/service/` leaves — `src/service/launchd.ts`, `src/service/systemd.ts`, `src/service/windows-ops.ts`, `src/service/windows-scheduler.ts`, `src/service/windows-taskxml.ts`, `src/service/state.ts`, `src/service/guards.ts`, `src/service/health.ts`, `src/service/repair.ts`, `src/service/orchestration.ts`, `src/service/diagnostics.ts`, `src/service/cli.ts`. Codex-home ownership accepts either the recorded path or the same existing physical directory so path aliases remain compatible across upgrades. Elevated Task Scheduler repair stages bounded payloads; the unelevated launcher pins every namespace ancestor and payload with non-reparse handles that deny write/delete sharing on the payload and delete sharing on each ancestor until UAC processing exits. | `src/cli/provider.ts` accepts the Google-only `--google-tool-schema-policy` creation flag and rejects an unknown value or non-Google effective adapter before persistence. The persisted field and default diff --git a/tests/codex-integration/codex-home-wsl.test.ts b/tests/codex-integration/codex-home-wsl.test.ts index 75733559782..119aaad7b4e 100644 --- a/tests/codex-integration/codex-home-wsl.test.ts +++ b/tests/codex-integration/codex-home-wsl.test.ts @@ -1,6 +1,6 @@ import { describe, expect, test } from "bun:test"; import { spawnSync } from "node:child_process"; -import { mkdirSync, mkdtempSync, realpathSync, writeFileSync } from "node:fs"; +import { mkdirSync, mkdtempSync, realpathSync, symlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { defaultCodexHome, wslAutomountRoot, listWslWindowsCodexHomes } from "../../src/codex/home"; @@ -212,4 +212,56 @@ describe("wsl.conf automount root", () => { env: { ...deps.env, CODEX_HOME: windowsCodexHome }, })).toBe(false); }); + + test("service ownership accepts an older lexical spelling of the current physical home", () => { + const lexicalHome = "/home/example/.codex"; + const physicalHome = "/srv/codex-home"; + const deps = { + env: {}, + homedir: () => "/home/example", + statSync: (() => ({ isDirectory: () => true })) as never, + realpathSync: (path: string) => path === lexicalHome ? physicalHome : path, + }; + + expect(serviceCodexHomeMatchesInstall(lexicalHome, deps)).toBe(true); + expect(serviceCodexHomeMatchesInstall("/srv/other-home", deps)).toBe(false); + }); + + // The injected realpath seam above isolates the policy; this exercises the production + // resolver itself — a real junction (Windows) or directory symlink spells the same + // physical home two ways. + test("service ownership accepts a real junction spelling of the physical home", () => { + const root = mkdtempSync(join(tmpdir(), "ocx-junction-home-")); + try { + const physical = join(root, "real-codex"); + const alias = join(root, "alias-codex"); + mkdirSync(physical, { recursive: true }); + symlinkSync(physical, alias, "junction"); + + const deps = { env: { CODEX_HOME: physical }, homedir: () => root }; + expect(serviceCodexHomeMatchesInstall(alias, deps)).toBe(true); + expect(serviceCodexHomeMatchesInstall(join(root, "other-codex"), deps)).toBe(false); + } finally { + removeTreeWithRetry(root); + } + }); + + // The home directory itself can sit under a junctioned ancestor — then the recorded + // spelling resolves through an intermediate link, not a link at the final component. + test("service ownership resolves a home through a junctioned parent directory", () => { + const root = mkdtempSync(join(tmpdir(), "ocx-junction-parent-")); + try { + const parentReal = join(root, "parent-real"); + const physical = join(parentReal, ".codex"); + mkdirSync(physical, { recursive: true }); + const parentAlias = join(root, "parent-alias"); + symlinkSync(parentReal, parentAlias, "junction"); + + const deps = { env: { CODEX_HOME: physical }, homedir: () => root }; + expect(serviceCodexHomeMatchesInstall(join(parentAlias, ".codex"), deps)).toBe(true); + expect(serviceCodexHomeMatchesInstall(join(parentAlias, "other"), deps)).toBe(false); + } finally { + removeTreeWithRetry(root); + } + }); }); diff --git a/tests/codex-integration/codex-service-manager-probe-hardening.test.ts b/tests/codex-integration/codex-service-manager-probe-hardening.test.ts index f468a8ff510..a7ece534269 100644 --- a/tests/codex-integration/codex-service-manager-probe-hardening.test.ts +++ b/tests/codex-integration/codex-service-manager-probe-hardening.test.ts @@ -552,6 +552,9 @@ describe("Windows ownership probe hardening regressions", () => { winswStatus: () => "nonexistent", statePaths: [statePath], currentHomes: { codexHome: currentCodexHome, opencodexHome: configDir }, + // Keep the foreign-vs-current comparison lexical: the fixture paths are + // intentionally not on disk, and a real ENOENT is "unknown", not "different". + realpathSync: (path: string) => path, }); expect(result.ownership).toBe("unknown"); @@ -687,6 +690,7 @@ describe("Windows ownership probe hardening regressions", () => { winswStatus: () => "started", statePaths: [statePath], currentHomes: { codexHome, opencodexHome: configDir }, + realpathSync: (path: string) => path, }); expect(result.ownership).toBe("unknown"); diff --git a/tests/codex-integration/codex-service-manager-probe.test.ts b/tests/codex-integration/codex-service-manager-probe.test.ts index 7d0dff0cbae..9fd445d5839 100644 --- a/tests/codex-integration/codex-service-manager-probe.test.ts +++ b/tests/codex-integration/codex-service-manager-probe.test.ts @@ -819,7 +819,7 @@ describe("ownership refuses what it cannot prove", () => { * homedir(), which no test sandbox moves. Left alone, these fixtures would * read the developer's real installation and call their own machine foreign. */ - function own(extra: { run: ProbeRunner }) { + function own(extra: { run: ProbeRunner; realpathSync?: (path: string) => string }) { const codexHome = join(home, ".codex"); const opencodexHome = join(home, ".opencodex"); return { @@ -861,7 +861,43 @@ describe("ownership refuses what it cannot prove", () => { const { opencodexHome } = useHomes(); writeState(opencodexHome, "/elsewhere/.codex", "/elsewhere/.opencodex"); const { run } = recorder(() => ({ status: 113 })); - expect(inspectNativeCodexOwnership(own({ run })).ownership).toBe("foreign"); + // Identity resolution keeps this comparison lexical so it proves "different", not "unknown". + const realpathSync = (path: string) => path; + expect(inspectNativeCodexOwnership(own({ run, realpathSync })).ownership).toBe("foreign"); + }); + + /* + * A realpath failure (EACCES, EPERM, a directory that vanished mid-compare, + * transient I/O) is not evidence the home is different. Collapsing it to + * "different" would make the unattended preflight report a definitive + * foreign install — and wrongly block stop, repair, uninstall, and native + * writes with incorrect recovery guidance — on nothing but an I/O hiccup. + */ + test("an unresolvable recorded home is unknown, not foreign", () => { + const { codexHome, opencodexHome } = useHomes(); + const recordedHome = join(home, "recorded-alias"); + writeState(opencodexHome, recordedHome, opencodexHome); + const { run } = recorder(() => ({ status: 113 })); + const realpathSync = (path: string) => { + if (path === recordedHome) throw Object.assign(new Error("access denied"), { code: "EACCES" }); + return path; + }; + + const result = inspectNativeCodexOwnership(own({ run, realpathSync })); + expect(result.ownership).toBe("unknown"); + expect(result.reason).toContain("could not be resolved"); + expect(result.reason).not.toContain("foreign"); + }); + + // An older install may have recorded a junction or symlink spelling of the + // home this process now knows canonically — same directory, different name. + test("state spelling the current home through an alias is owned", () => { + const { codexHome, opencodexHome } = useHomes(); + const aliasHome = join(home, "codex-alias"); + writeState(opencodexHome, aliasHome, opencodexHome); + const { run } = recorder(() => ({ status: 113 })); + const realpathSync = (path: string) => path === aliasHome ? codexHome : path; + expect(inspectNativeCodexOwnership(own({ run, realpathSync })).ownership).toBe("owned"); }); /* @@ -876,7 +912,10 @@ describe("ownership refuses what it cannot prove", () => { writePlist("/elsewhere/.codex", "/elsewhere/.opencodex"); const { run } = recorder(() => ({ status: 113 })); - const result = inspectNativeCodexOwnership(own({ run })); + // Identity resolution keeps the claim comparison lexical: the fixture + // intends a genuinely different home, not an unresolvable one. + const realpathSync = (path: string) => path; + const result = inspectNativeCodexOwnership(own({ run, realpathSync })); expect(result.ownership).toBe("unknown"); expect(result.reason).toContain("different homes"); }); diff --git a/tests/service/service-sqlite-home.test.ts b/tests/service/service-sqlite-home.test.ts index 17af32d6673..faceb389596 100644 --- a/tests/service/service-sqlite-home.test.ts +++ b/tests/service/service-sqlite-home.test.ts @@ -75,6 +75,28 @@ describe("service install state Codex SQLite home binding", () => { expect(() => assertServiceEnvironmentMatchesInstall()).toThrow("Codex SQLite home"); }); + test("accepts a recorded SQLite home that names the same physical directory through an alias", () => { + const codexHome = join(TEST_DIR, "codex-home"); + const sqliteHome = join(TEST_DIR, "sqlite-home"); + const sqliteAlias = join(TEST_DIR, "sqlite-alias"); + process.env.CODEX_HOME = codexHome; + process.env.CODEX_SQLITE_HOME = sqliteHome; + writeInstallState({ + version: 2, + codexHome, + codexSqliteHome: sqliteAlias, + opencodexHome: TEST_DIR, + backend: "scheduler", + }); + + const realpathSync = (path: string) => path === sqliteAlias ? sqliteHome : path; + expect(() => assertServiceEnvironmentMatchesInstall({ realpathSync })).not.toThrow(); + + const otherHome = join(TEST_DIR, "other-sqlite-home"); + const divergentRealpath = (path: string) => path === sqliteAlias ? otherHome : path; + expect(() => assertServiceEnvironmentMatchesInstall({ realpathSync: divergentRealpath })).toThrow("Codex SQLite home"); + }); + test("parses codexSqliteHome and rejects an empty value", () => { const valid = { version: 2, diff --git a/tests/service/service-wsl-home-ownership.test.ts b/tests/service/service-wsl-home-ownership.test.ts index 1d5ec1a2342..50c989ce242 100644 --- a/tests/service/service-wsl-home-ownership.test.ts +++ b/tests/service/service-wsl-home-ownership.test.ts @@ -61,6 +61,7 @@ describe("WSL service ownership after Windows home discovery", () => { expect(inspectNativeCodexOwnership({ statePaths: [statePath], currentHomes: { codexHome: windowsHome, opencodexHome: root }, + realpathSync: deps.realpathSync, }).ownership).toBe("foreign"); }); From 43dbcee26569bb65606ab8bbee4b38b3e760773e Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:43:43 +0900 Subject: [PATCH 24/75] fix(service): keep a vanished recorded home foreign Follow-up to #6036. Its tri-state comparison reported any realpath failure as unknown, so a differently spelled recorded home that no longer exists stopped refusing ocx restore, which fails the WP13 composed acceptance contract on dev. A missing path cannot be an alias of the current home, so ENOENT and ENOTDIR compare different; EACCES and other unproven errors stay unknown. --- src/service/state.ts | 12 +++++++++++- .../codex-service-manager-probe.test.ts | 14 ++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/src/service/state.ts b/src/service/state.ts index d4f390773e4..ff93d70f573 100644 --- a/src/service/state.ts +++ b/src/service/state.ts @@ -869,11 +869,21 @@ export type ServicePathComparison = "same" | "different" | "unknown"; export function compareServicePathToInstall(recorded: string, current: string, deps: CodexHomeDeps = {}): ServicePathComparison { if (serviceHomeMatches(recorded, current)) return "same"; const realpath = deps.realpathSync ?? realpathSync; + let currentPhysical: string; try { - return serviceHomeMatches(realpath(recorded), realpath(current)) ? "same" : "different"; + currentPhysical = realpath(current); } catch { return "unknown"; } + try { + return serviceHomeMatches(realpath(recorded), currentPhysical) ? "same" : "different"; + } catch (error) { + // The spellings already differ. A recorded path that no longer exists cannot be an alias of + // the current home, so it stays a mismatch (a stale mount keeps the foreign-owner refusal); + // only an error that leaves existence unproven, such as EACCES, is indeterminate. + const code = (error as NodeJS.ErrnoException | null)?.code; + return code === "ENOENT" || code === "ENOTDIR" ? "different" : "unknown"; + } } /** Lexical compare first; when spellings differ, compare the directories both resolve to so a diff --git a/tests/codex-integration/codex-service-manager-probe.test.ts b/tests/codex-integration/codex-service-manager-probe.test.ts index 9fd445d5839..6c2281d5ea5 100644 --- a/tests/codex-integration/codex-service-manager-probe.test.ts +++ b/tests/codex-integration/codex-service-manager-probe.test.ts @@ -889,6 +889,20 @@ describe("ownership refuses what it cannot prove", () => { expect(result.reason).not.toContain("foreign"); }); + // A differently spelled recorded home that no longer exists cannot be an alias of the current + // home. It stays foreign, so a stale mount keeps refusing restore and startup writes. + test("a vanished, differently spelled recorded home stays foreign", () => { + const { opencodexHome } = useHomes(); + const recordedHome = join(home, "unmounted-home"); + writeState(opencodexHome, recordedHome, opencodexHome); + const { run } = recorder(() => ({ status: 113 })); + const realpathSync = (path: string) => { + if (path === recordedHome) throw Object.assign(new Error("absent"), { code: "ENOENT" }); + return path; + }; + expect(inspectNativeCodexOwnership(own({ run, realpathSync })).ownership).toBe("foreign"); + }); + // An older install may have recorded a junction or symlink spelling of the // home this process now knows canonically — same directory, different name. test("state spelling the current home through an alias is owned", () => { From 642cb61df11695c4786c23b9cf1cfb46137f08f9 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 14:46:45 +0900 Subject: [PATCH 25/75] fix(search): bind Devin OAuth to routed provider (#6038) Carried from #6038 into merge train round 3. Co-authored-by: Epinephrine --- src/oauth/index.ts | 24 +++-- src/server/search.ts | 2 +- src/web-search/devin-executor.ts | 2 +- structure/runtime.md | 2 +- .../server/api-key-scope-alpha-search.test.ts | 98 +++++++++++++++++++ 5 files changed, 116 insertions(+), 12 deletions(-) diff --git a/src/oauth/index.ts b/src/oauth/index.ts index 2d14dfd5421..f705bf43df2 100644 --- a/src/oauth/index.ts +++ b/src/oauth/index.ts @@ -466,7 +466,7 @@ export function publicOAuthAuthenticationErrorMessage(error: unknown): string { return "OAuth authentication failed. Check the OpenCodex account status and retry."; } -function accessSnapshot(provider: string, accountId: string, cred: OAuthCredentials): OAuthAccessSnapshot { +function accessSnapshot(provider: string, accountId: string, cred: OAuthCredentials, oauthProvider = provider): OAuthAccessSnapshot { // Derived, not read back: a stored `authType` is trusted when present, but a credential imported // before the field existed still routes correctly because the client pair implies SSO OIDC. const kiroAuthType = cred.kiro?.authType @@ -484,9 +484,11 @@ function accessSnapshot(provider: string, accountId: string, cred: OAuthCredenti // Validated here, not at the call site: an unvalidated origin from a legacy or crafted // credential must never travel with a bearer, and dropping it makes the transport fall back to // the canonical host rather than to whatever the previous account was using. - const accountApiBaseUrl = provider === "github-copilot" + // The host rides on the OAuth definition the snapshot was resolved through, not the routed + // slot name: a custom provider reusing the Devin definition keeps its stored tenant URL. + const accountApiBaseUrl = oauthProvider === "github-copilot" ? validateCopilotApiBaseUrl(cred.apiBaseUrl) - : provider === "devin" || provider === "devin-cli" + : oauthProvider === "devin" || oauthProvider === "devin-cli" ? validateDevinApiBaseUrl(cred.apiBaseUrl) : undefined; return { @@ -550,9 +552,10 @@ async function resolveAccessSnapshotForAccount( accountId: string, rejectedGeneration?: string, requireUsableAccount = false, + oauthProvider = provider, ): Promise { - const def = OAUTH_PROVIDERS[provider]; - if (!def) throw new UnsupportedOAuthProviderError(provider); + const def = OAUTH_PROVIDERS[oauthProvider]; + if (!def) throw new UnsupportedOAuthProviderError(oauthProvider); // One store read answers both questions. A caller that opts in gets the account REJECTED // when it needs reauthentication, which a bare credential read cannot detect: a revoked // account keeps a readable credential, so resolution would otherwise succeed and the @@ -561,7 +564,7 @@ async function resolveAccessSnapshotForAccount( if (!row) throw new OAuthLoginRequiredError(provider); if (requireUsableAccount && row.needsReauth) throw new OAuthLoginRequiredError(provider); const cred = row.credential; - const current = accessSnapshot(provider, accountId, cred); + const current = accessSnapshot(provider, accountId, cred, oauthProvider); if (rejectedGeneration !== undefined && current.generation !== rejectedGeneration) return current; if (rejectedGeneration === undefined && cred.expires > Date.now() + REFRESH_SKEW_MS) return current; @@ -595,7 +598,7 @@ async function resolveAccessSnapshotForAccount( if (persisted.access !== accessToken) { throw new Error(`OAuth refresh persisted an unexpected access token for ${provider}`); } - return accessSnapshot(provider, accountId, persisted); + return accessSnapshot(provider, accountId, persisted, oauthProvider); })().catch(error => { if (abort.signal.reason instanceof OAuthTokenRefreshStaleError) throw abort.signal.reason; throw error; @@ -607,10 +610,13 @@ async function resolveAccessSnapshotForAccount( return refresh; } -export async function getValidAccessTokenSnapshot(provider: string): Promise { +export async function getValidAccessTokenSnapshot( + provider: string, + options: { oauthProvider?: string } = {}, +): Promise { const set = getAccountSet(provider); if (!set) throw new OAuthLoginRequiredError(provider); - return resolveAccessSnapshotForAccount(provider, set.activeAccountId); + return resolveAccessSnapshotForAccount(provider, set.activeAccountId, undefined, false, options.oauthProvider); } /** Providers whose upstream-401 replay path may force a snapshot refresh. */ diff --git a/src/server/search.ts b/src/server/search.ts index 60db8c25ab3..33d52c71528 100644 --- a/src/server/search.ts +++ b/src/server/search.ts @@ -84,7 +84,7 @@ export async function handleSearch( logCtx.routeDecision = route.routeDecision; return handleDevinAlphaSearch( body, - "devin", + route.providerName, config.search?.timeoutMs ?? SEARCH_UPSTREAM_TIMEOUT_MS, req.signal, ); diff --git a/src/web-search/devin-executor.ts b/src/web-search/devin-executor.ts index dde633c89f5..7da0fac1748 100644 --- a/src/web-search/devin-executor.ts +++ b/src/web-search/devin-executor.ts @@ -126,7 +126,7 @@ export async function resolveDevinWebSearchSnapshot( try { const selection = captureOAuthAccountSelection(credentialProvider); if (!selection) return { error: "devin web search auth failed: no signed-in account" }; - const snapshot = await getValidAccessTokenSnapshot(credentialProvider); + const snapshot = await getValidAccessTokenSnapshot(credentialProvider, { oauthProvider: "devin" }); const committed = await commitOAuthAccountSelection(credentialProvider, snapshot.accountId, { expectedSelection: selection, expectedCredentialGeneration: snapshot.generation, diff --git a/structure/runtime.md b/structure/runtime.md index 0d26fe16cf3..3fd46e1ca2d 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -448,7 +448,7 @@ following a final symlink, so an exchange during a mutation cannot redirect the Config JSON preserves the boolean; only literal true activates the role-changing transform. Claude skill-bundle marker parsing follows the [bounded inbound contract](data-planes/inbound-compat.md#claude-skill-marker-path-bound). The lightweight top-level CLI help counts Cline CLI among the fifteen registered export clients; registry parity remains covered by the client help and integration tests. -Devin CLI credential path composition in `src/oauth/devin/cli-import.ts` follows the selected platform: Windows uses Win32 APPDATA paths, other platforms use POSIX XDG-data paths. The explicit absolute override remains verbatim; credential parsing and login behavior are unchanged. The `src/providers/devin-provider-merge-migration.ts` startup migration treats the legacy provider row and its OAuth slot as one account-bound unit: an occupied destination or a refused config projection leaves both unchanged, and both backups complete before either file changes. The adapter takes a tenant host only from the stored account that owns the exact key being transmitted, in the literal slot or, during a detached rekey window, the alias slot, so separately configured or forwarded credentials and non-owning accounts cannot lend another account's destination. +Devin CLI credential path composition in `src/oauth/devin/cli-import.ts` follows the selected platform: Windows uses Win32 APPDATA paths, other platforms use POSIX XDG-data paths. The explicit absolute override remains verbatim; credential parsing and login behavior are unchanged. The `src/providers/devin-provider-merge-migration.ts` startup migration treats the legacy provider row and its OAuth slot as one account-bound unit: an occupied destination or a refused config projection leaves both unchanged, and both backups complete before either file changes. The adapter takes a tenant host only from the stored account that owns the exact key being transmitted, in the literal slot or, during a detached rekey window, the alias slot, so separately configured or forwarded credentials and non-owning accounts cannot lend another account's destination. Native Devin alpha search likewise resolves OAuth from the routed provider name that admission checked, rather than borrowing the canonical `devin` slot for a custom Devin-adapter row. Native Chat applies qualifying effort ceilings independently of model pins; pin selection precedes the cap and only pins or cap rewrites enter wire mapping. The [catalog effort contract](catalog.md#ultra-reasoning-level) records the V1/compaction exemptions and caller-preservation boundary. Pool quota producers and account commands follow the [bounded raw-observation contract](providers/openai-accounts.md#bounded-pool-quota-observations), separate from the latest display snapshot and capacity estimates. diff --git a/tests/server/api-key-scope-alpha-search.test.ts b/tests/server/api-key-scope-alpha-search.test.ts index 769d972a130..beb4103298b 100644 --- a/tests/server/api-key-scope-alpha-search.test.ts +++ b/tests/server/api-key-scope-alpha-search.test.ts @@ -10,7 +10,9 @@ import { afterEach, beforeEach, expect, test } from "bun:test"; import { existsSync, mkdirSync } from "node:fs"; import { join } from "node:path"; +import { encodeMessage, encodeString } from "../../src/adapters/devin/cloud-direct/wire"; import { saveConfig } from "../../src/config"; +import { saveCredential } from "../../src/oauth/store"; import { MODEL_NOT_ALLOWED_FOR_KEY, UNNAMED_DESTINATION_MODEL } from "../../src/server/admission-model-scope"; import type { DataPlaneAdmission } from "../../src/server/auth-cors"; import type { RequestLogContext } from "../../src/server/request-log"; @@ -199,6 +201,102 @@ test("the relay still runs when the scope names its destination", async () => { expect(upstreamCalls[0]).toContain("/alpha/search"); }); +test("a scoped custom Devin route spends only that provider's OAuth credential", async () => { + const customToken = "team-devin-token"; + await saveCredential("team-devin", { + access: customToken, + refresh: customToken, + expires: Number.MAX_SAFE_INTEGER, + apiBaseUrl: "https://team-devin.example", + }); + await saveCredential("devin", { + access: "canonical-devin-token", + refresh: "canonical-devin-token", + expires: Number.MAX_SAFE_INTEGER, + apiBaseUrl: "https://canonical-devin.example", + }); + let requestBody = Buffer.alloc(0); + globalThis.fetch = (async (input: RequestInfo | URL, init?: RequestInit) => { + upstreamCalls.push(String(input)); + requestBody = Buffer.from(init?.body as Uint8Array); + const result = Buffer.concat([ + encodeString(3, "https://example.test"), + encodeString(4, "Result"), + ]); + return new Response(encodeMessage(1, result), { status: 200 }); + }) as typeof fetch; + const config = { + port: 0, + defaultProvider: "team-devin", + providers: { + "team-devin": { adapter: "devin", authMode: "oauth", baseUrl: "https://server.codeium.com" }, + }, + apiKeys: keys({ allowedProviders: ["team-devin"] }), + } as OcxConfig; + + const response = await handleSearch( + sidecarRequest(searchBody("team-devin/swe-2")), + config, + logContext(), + undefined, + SCOPED, + ); + expect(response.status).toBe(200); + expect(upstreamCalls).toEqual([ + "https://server.codeium.com/exa.api_server_pb.ApiServerService/GetWebSearchResults", + ]); + expect(requestBody.includes(Buffer.from(customToken))).toBe(true); + expect(requestBody.includes(Buffer.from("canonical-devin-token"))).toBe(false); +}); + +test("a custom Devin route searches the tenant its credential names", async () => { + const customToken = "team-devin-token"; + await saveCredential("team-devin", { + access: customToken, + refresh: customToken, + expires: Number.MAX_SAFE_INTEGER, + apiBaseUrl: "https://eu.windsurf.com/_route/api_server", + }); + await saveCredential("devin", { + access: "canonical-devin-token", + refresh: "canonical-devin-token", + expires: Number.MAX_SAFE_INTEGER, + apiBaseUrl: "https://canonical-devin.example", + }); + let requestBody = Buffer.alloc(0); + globalThis.fetch = (async (input: RequestInfo | URL, init?: RequestInit) => { + upstreamCalls.push(String(input)); + requestBody = Buffer.from(init?.body as Uint8Array); + const result = Buffer.concat([ + encodeString(3, "https://example.test"), + encodeString(4, "Result"), + ]); + return new Response(encodeMessage(1, result), { status: 200 }); + }) as typeof fetch; + const config = { + port: 0, + defaultProvider: "team-devin", + providers: { + "team-devin": { adapter: "devin", authMode: "oauth", baseUrl: "https://server.codeium.com" }, + }, + apiKeys: keys({ allowedProviders: ["team-devin"] }), + } as OcxConfig; + + const response = await handleSearch( + sidecarRequest(searchBody("team-devin/swe-2")), + config, + logContext(), + undefined, + SCOPED, + ); + expect(response.status).toBe(200); + expect(upstreamCalls).toEqual([ + "https://eu.windsurf.com/_route/api_server/exa.api_server_pb.ApiServerService/GetWebSearchResults", + ]); + expect(requestBody.includes(Buffer.from(customToken))).toBe(true); + expect(requestBody.includes(Buffer.from("canonical-devin-token"))).toBe(false); +}); + test("the sidecar fallback refuses the backend it would have spent", async () => { const response = await handleSearch( sidecarRequest(searchBody(SEARCH_MODEL)), From 2849da26d683e0d80d09118542512be0aa9db480 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 14:46:57 +0900 Subject: [PATCH 26/75] fix(update): isolate pnpm probes and mutation workspaces (#6048) Carried from #6048 into merge train round 3. Co-authored-by: Epinephrine --- bin/ocx.mjs | 42 +++++---- scripts/test-layout/layout.json | 1 + src/update/async-check.ts | 6 +- src/update/index.ts | 42 ++++++--- src/update/pnpm-read-policy.d.mts | 4 + src/update/pnpm-read-policy.mjs | 44 +++++++++ structure/ops/service-and-sidecars.md | 2 +- tests/fixtures/test-layout-expected.json | 1 + tests/update/pnpm-command-isolation.test.ts | 57 ++++++++++++ tests/update/update-refresh.test.ts | 98 ++++++++++++++++++++- 10 files changed, 265 insertions(+), 32 deletions(-) create mode 100644 src/update/pnpm-read-policy.d.mts create mode 100644 src/update/pnpm-read-policy.mjs create mode 100644 tests/update/pnpm-command-isolation.test.ts diff --git a/bin/ocx.mjs b/bin/ocx.mjs index ee7e0494b6e..406748d8e1e 100755 --- a/bin/ocx.mjs +++ b/bin/ocx.mjs @@ -42,6 +42,7 @@ import { resolvePnpmGlobalOwner, runPnpmGlobalUpdate, } from "../src/update/pnpm-global-install.mjs"; +import { PNPM_READ_CWD, withPnpmCommandCwd, pnpmReadEnvironment } from "../src/update/pnpm-read-policy.mjs"; import { checkRegistryPackageIntegrity } from "../src/update/registry-integrity.mjs"; import { hasPendingTeardownIn } from "../src/config/pending-teardown-names.mjs"; import { @@ -208,6 +209,8 @@ function runPackageManagerSelfUpdate(manager) { encoding: "utf8", timeout: 20_000, windowsHide: true, + cwd: PNPM_READ_CWD, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(process.env)), ...invocation.options, }); }, @@ -221,6 +224,20 @@ function runPackageManagerSelfUpdate(manager) { const managerInvocation = args => manager === "pnpm" ? pnpmOwnerInvocation(owner, args) : npmInvocation(args); + // Read-only pnpm probes run from the installed package directory with project pnpmfiles + // disabled, so an attacker-controlled cwd cannot execute hooks during the update check. + const readProbeOptions = invocation => ({ + encoding: "utf8", + timeout: 12000, + windowsHide: true, + ...(manager === "pnpm" + ? { + cwd: PNPM_READ_CWD, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(invocation.env ?? process.env)), + } + : invocation.env ? { env: invocation.env } : {}), + ...invocation.options, + }); const latestInvocation = managerInvocation(["view", `${PKG}@${tag}`, "version"]); const installArgs = manager === "pnpm" ? ["add", "-g", "--allow-build=bun", `${PKG}@${tag}`] @@ -230,13 +247,7 @@ function runPackageManagerSelfUpdate(manager) { console.error(`opencodex: could not resolve ${manager} from a trusted absolute PATH entry; aborting before stopping the proxy.`); process.exit(1); } - const latestResult = spawnSync(latestInvocation.file, latestInvocation.args, { - encoding: "utf8", - timeout: 12000, - windowsHide: true, - ...(latestInvocation.env ? { env: latestInvocation.env } : {}), - ...latestInvocation.options, - }); + const latestResult = spawnSync(latestInvocation.file, latestInvocation.args, readProbeOptions(latestInvocation)); const latest = latestResult.status === 0 && typeof latestResult.stdout === "string" ? latestResult.stdout.trim() : ""; console.log(`opencodex v${current} (installed via ${manager}, tag ${tag})`); @@ -248,13 +259,7 @@ function runPackageManagerSelfUpdate(manager) { const integrity = checkRegistryPackageIntegrity(PKG, latest || null, args => { const invocation = managerInvocation(args); if (!invocation) return { status: 1 }; - return spawnSync(invocation.file, invocation.args, { - encoding: "utf8", - timeout: 12000, - windowsHide: true, - ...(invocation.env ? { env: invocation.env } : {}), - ...invocation.options, - }); + return spawnSync(invocation.file, invocation.args, readProbeOptions(invocation)); }); if (integrity.ok === false) { console.error(`opencodex: ${integrity.reason}; aborting before stopping the proxy.`); @@ -768,14 +773,17 @@ function runPackageManagerSelfUpdate(manager) { runPnpm: (args, capture = false) => { const invocation = pnpmOwnerInvocation(owner, args); if (!invocation) return { status: 1 }; - return spawnSync(invocation.file, invocation.args, { + return withPnpmCommandCwd(args, cwd => spawnSync(invocation.file, invocation.args, { ...invocation.options, stdio: capture ? "pipe" : "inherit", encoding: "utf8", timeout: 180000, windowsHide: true, - env: unprivilegedOwnershipMutationEnvironment(invocation.env ?? process.env), - }); + // Reads probe from the package dir; mutations (add -g, rollback) must not + // keep a cwd handle inside the package Windows is replacing. + cwd, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(invocation.env ?? process.env)), + })); }, log: line => console.log(line), }); diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 4856cd23048..331193e92cb 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -168,6 +168,7 @@ } }, "explicit": { + "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", "responses-compaction-recovery-policy.test.ts": "responses", "plugin-loader.test.ts": "lib", "plugin-upstream-hooks.test.ts": "lib", "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", diff --git a/src/update/async-check.ts b/src/update/async-check.ts index e24a6eccc19..08ffa5fb3a2 100644 --- a/src/update/async-check.ts +++ b/src/update/async-check.ts @@ -2,6 +2,7 @@ import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process"; import { unprivilegedOwnershipMutationEnvironment } from "../service/ownership-mutation-lease.mjs"; import { PKG, registrySpawnTarget, type Channel, type Installer } from "./index"; import type { PnpmGlobalOwner } from "./pnpm-global-install.mjs"; +import { PNPM_READ_CWD, pnpmReadEnvironment } from "./pnpm-read-policy.mjs"; export const REGISTRY_DEADLINE_MS = 12_000; export const REGISTRY_OUTPUT_LIMIT = 4_096; @@ -64,7 +65,10 @@ export async function latestVersionAsync( child = deps.spawnFn(target.bin, target.args, { stdio: ["pipe", "pipe", "pipe"], windowsHide: true, - env: unprivilegedOwnershipMutationEnvironment(target.env ?? process.env), + cwd: installer === "pnpm" ? PNPM_READ_CWD : undefined, + env: installer === "pnpm" + ? pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(target.env ?? process.env)) + : unprivilegedOwnershipMutationEnvironment(target.env ?? process.env), ...target.options, }) as ChildProcessWithoutNullStreams; } catch { diff --git a/src/update/index.ts b/src/update/index.ts index 098dcde03fd..1f793952ed6 100644 --- a/src/update/index.ts +++ b/src/update/index.ts @@ -41,6 +41,7 @@ import { withProcessRuntimeProvenance } from "../lib/bun-runtime"; import { withoutSiblingMarker } from "../codex/sibling-start"; import { packageVersion } from "../lib/package-version"; import { selfLaunchArgv } from "../lib/self-launch-argv"; +import { PNPM_READ_CWD, withPnpmCommandCwd, pnpmReadEnvironment } from "./pnpm-read-policy.mjs"; /** * A `codex-history-backup-*.json` surviving a stop means the native-history restore was @@ -98,27 +99,33 @@ function runPnpmCandidate( commandPath: string, args: readonly string[], capture = false, + spawn: typeof spawnSync = spawnSync, ): { status: number | null; stdout?: string | null; stderr?: string | null } { const invocation = pnpmInvocationForPath(commandPath, args); if (!invocation) return { status: 1 }; - return spawnSync(invocation.file, invocation.args, { + return spawn(invocation.file, invocation.args, { stdio: capture ? "pipe" : "ignore", encoding: "utf8", timeout: 20_000, windowsHide: true, - env: unprivilegedOwnershipMutationEnvironment(process.env), + cwd: PNPM_READ_CWD, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(process.env)), ...invocation.options, }); } /** Resolve the exact pnpm executable/group/bin that own this package. */ -export function resolveCurrentPnpmGlobalOwner(invoked = process.argv[1]): PnpmGlobalOwnerResult { +export function resolveCurrentPnpmGlobalOwner( + invoked = process.argv[1], + deps: { commandPaths?: readonly string[]; spawn?: typeof spawnSync } = {}, +): PnpmGlobalOwnerResult { + const spawn = deps.spawn ?? spawnSync; return resolvePnpmGlobalOwner({ packageName: PKG, packagePath: packageRoot(), - commandPaths: resolvePnpmCommands(), + commandPaths: deps.commandPaths ?? resolvePnpmCommands(), runningShimPath: runningPnpmShimPath(invoked), - runPnpm: runPnpmCandidate, + runPnpm: (commandPath, args, capture) => runPnpmCandidate(commandPath, args, capture, spawn), }); } @@ -136,22 +143,26 @@ function ownerPnpmTarget( }; } -function runOwnedPnpm( +export function runOwnedPnpm( owner: PnpmGlobalOwner, args: readonly string[], capture: boolean, stdio: "inherit" | "pipe" | "ignore" = capture ? "pipe" : "inherit", + spawn: typeof spawnSync = spawnSync, ): { status: number | null; stdout?: string | null; stderr?: string | null } { const target = ownerPnpmTarget(owner, args); if (!target) return { status: 1 }; - return spawnSync(target.bin, target.args, { + return withPnpmCommandCwd(args, cwd => spawn(target.bin, target.args, { stdio, encoding: "utf8", timeout: 180_000, windowsHide: true, - env: unprivilegedOwnershipMutationEnvironment(target.env), + // Reads probe from the package dir; `add -g`/rollback children run from a neutral + // directory so a Windows cwd handle never pins open the package pnpm is replacing. + cwd, + env: pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(target.env)), ...target.options, - }); + })); } /** Re-read the owning group's active package and return its verified launcher. */ @@ -276,16 +287,20 @@ export function latestVersion( tag: string, installer: Installer = detectInstall(), owner?: PnpmGlobalOwner, + spawn: typeof spawnSync = spawnSync, ): string | null { const resolvedOwner = installer === "pnpm" ? selectedPnpmOwner(owner) : undefined; if (installer === "pnpm" && !resolvedOwner) return null; const manager = registrySpawnTarget(installer, ["view", `${PKG}@${tag}`, "version"], resolvedOwner); if (!manager) return null; - const r = spawnSync(manager.bin, manager.args, { + const r = spawn(manager.bin, manager.args, { encoding: "utf8", timeout: 12000, windowsHide: true, - env: unprivilegedOwnershipMutationEnvironment(manager.env ?? process.env), + cwd: installer === "pnpm" ? PNPM_READ_CWD : undefined, + env: installer === "pnpm" + ? pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(manager.env ?? process.env)) + : unprivilegedOwnershipMutationEnvironment(manager.env ?? process.env), ...manager.options, }); return r.status === 0 && typeof r.stdout === "string" ? (r.stdout.trim() || null) : null; @@ -342,7 +357,10 @@ export function checkUpdatePackageIntegrity( encoding: "utf8", timeout: 12000, windowsHide: true, - env: unprivilegedOwnershipMutationEnvironment(target.env ?? process.env), + cwd: installer === "pnpm" ? PNPM_READ_CWD : undefined, + env: installer === "pnpm" + ? pnpmReadEnvironment(unprivilegedOwnershipMutationEnvironment(target.env ?? process.env)) + : unprivilegedOwnershipMutationEnvironment(target.env ?? process.env), ...target.options, }); }); diff --git a/src/update/pnpm-read-policy.d.mts b/src/update/pnpm-read-policy.d.mts new file mode 100644 index 00000000000..f2bf6733ea0 --- /dev/null +++ b/src/update/pnpm-read-policy.d.mts @@ -0,0 +1,4 @@ +export declare const PNPM_READ_CWD: string; +/** The callback must complete synchronously before its temporary workspace is cleaned. */ +export declare function withPnpmCommandCwd(args: readonly string[], run: (cwd: string) => T): T; +export declare function pnpmReadEnvironment(env?: Record): Record; diff --git a/src/update/pnpm-read-policy.mjs b/src/update/pnpm-read-policy.mjs new file mode 100644 index 00000000000..97940fa1cf2 --- /dev/null +++ b/src/update/pnpm-read-policy.mjs @@ -0,0 +1,44 @@ +import { mkdtempSync, lstatSync, writeFileSync, unlinkSync, rmdirSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; + +/** Read probes never inherit the caller's project as their working directory. */ +export const PNPM_READ_CWD = dirname(fileURLToPath(import.meta.url)); +const MUTATIONS = new Set(["add", "install", "update", "remove", "uninstall"]); + +/** Run a synchronous pnpm mutation outside the package, in a private workspace boundary. */ +export function withPnpmCommandCwd(args, run) { + if (!MUTATIONS.has(args?.[0])) return run(PNPM_READ_CWD); + const cwd = mkdtempSync(join(tmpdir(), "ocx-pnpm-command-")); + const created = lstatSync(cwd); + const files = ["pnpm-workspace.yaml", ".npmrc"]; + try { + // Stop discovery at our own workspace, rather than inheriting a shared /tmp workspace. + writeFileSync(join(cwd, files[0]), "packages: []\nignorePnpmfile: true\n", { flag: "wx", mode: 0o600 }); + writeFileSync(join(cwd, files[1]), "ignore-pnpmfile=true\n", { flag: "wx", mode: 0o600 }); + return run(cwd); + } finally { + try { + const now = lstatSync(cwd); + if (!now.isDirectory() || now.isSymbolicLink() || now.dev !== created.dev || now.ino !== created.ino) { + throw new Error("temporary directory identity changed"); + } + // Never recursively remove pnpm-created or replacement contents. + for (const name of files) { + try { unlinkSync(join(cwd, name)); } + catch (error) { if (error?.code !== "ENOENT") throw error; } + } + rmdirSync(cwd); + } catch { + console.warn("[opencodex] Temporary pnpm workspace cleanup was incomplete; retained for inspection."); + } + } +} + +/** pnpm 11 uses pnpm_config_ while earlier versions use npm_config_. */ +export function pnpmReadEnvironment(env = process.env) { + const ignored = new Set(["npm_config_ignore_pnpmfile", "pnpm_config_ignore_pnpmfile"]); + const isolated = Object.fromEntries(Object.entries(env).filter(([key]) => !ignored.has(key.toLowerCase()))); + return { ...isolated, npm_config_ignore_pnpmfile: "true", pnpm_config_ignore_pnpmfile: "true" }; +} diff --git a/structure/ops/service-and-sidecars.md b/structure/ops/service-and-sidecars.md index 71163ae6387..d8702d82b46 100644 --- a/structure/ops/service-and-sidecars.md +++ b/structure/ops/service-and-sidecars.md @@ -328,7 +328,7 @@ detached replacement's environment drops the marker. Coverage: src/update/refresh-scheduler.ts owns the package cache timer and per-channel singleflight for the running proxy. Eligible npm, pnpm and Bun installs refresh missing or 20-hour-stale `version.json` after bind, check staleness hourly and retry failures with bounded backoff. Each server start owns one scheduler reference; the last matching stop disarms the timer. A stopped automatic lookup cannot write a late result, but an explicit check joining that lookup marks explicit interest and writes its successful result even if the last listener stops before it resolves. Source/mise installs and `OCX_DISABLE_UPDATE_CHECK=1` do not start automatic lookup; explicit requests remain available. -src/update/async-check.ts uses the existing owner-bound registry target with a bounded asynchronous child; pnpm owner discovery runs in src/update/pnpm-owner-worker.ts off the request loop. `src/update/notify.ts` writes successful results atomically and preserves a dismissal only for the same channel and version. The interactive pre-bind prompt reads the cache and does not launch a second detached refresh. `src/update/badge.ts` only reads the cache and reports unknown at 40 hours. +src/update/async-check.ts uses the existing owner-bound registry target with a bounded asynchronous child; pnpm owner discovery runs in src/update/pnpm-owner-worker.ts off the request loop. Read-only pnpm owner and registry probes — in the scheduler, the synchronous updater, and the `bin/ocx.mjs` package-manager self-update — run from the installed update module directory via `src/update/pnpm-read-policy.mjs` with project pnpmfiles disabled, never from the caller's workspace. pnpm mutations (`add -g`, rollback) instead run in unique private temporary workspaces outside the package, with an explicit empty workspace boundary to stop parent-project discovery. Both npm_config_ and pnpm_config_ ignore-pnpmfile controls are set case-insensitively for pnpm 10/11. Cleanup removes only known files and an empty unchanged directory; unexpected contents remain for inspection. On Windows, this also avoids pinning the replaced package as cwd. `src/update/notify.ts` writes successful results atomically and preserves a dismissal only for the same channel and version. The interactive pre-bind prompt reads the cache and does not launch a second detached refresh. `src/update/badge.ts` only reads the cache and reports unknown at 40 hours. The desktop badge snapshot in src/update/desktop-badge.ts is process-local display state keyed by a Tauri session id. A 60-second shell heartbeat renews receipt time; entries expire after 180 seconds and the store retains at most 32 sessions. It is separate from the package version cache and from the updater job/ownership transaction. A proxy restart reports unknown until a bound desktop shell republishes; no update installation can be authorized by this snapshot. diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index d8bfa12451f..903440aa89d 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1,4 +1,5 @@ { + "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", diff --git a/tests/update/pnpm-command-isolation.test.ts b/tests/update/pnpm-command-isolation.test.ts new file mode 100644 index 00000000000..207822878ce --- /dev/null +++ b/tests/update/pnpm-command-isolation.test.ts @@ -0,0 +1,57 @@ +import { expect, test } from "bun:test"; +import { existsSync, readFileSync, lstatSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { runOwnedPnpm } from "../../src/update/index"; +import { runPnpmGlobalUpdate } from "../../src/update/pnpm-global-install.mjs"; +import { PNPM_READ_CWD, pnpmReadEnvironment, withPnpmCommandCwd } from "../../src/update/pnpm-read-policy.mjs"; + +test("both pnpm environment prefixes override every case variant without mutating the parent", () => { + const original = { npm_config_ignore_pnpmfile: "false", NPM_CONFIG_IGNORE_PNPMFILE: "false", PnPm_CoNfIg_IgNoRe_PnPmFiLe: "false", pnpm_config_ignore_pnpmfile: "false", KEEP: "retained" }; + const result = pnpmReadEnvironment(original); + expect(result).toEqual({ npm_config_ignore_pnpmfile: "true", pnpm_config_ignore_pnpmfile: "true", KEEP: "retained" }); + expect(original.npm_config_ignore_pnpmfile).toBe("false"); +}); + +test("mutation workspaces are unique, bounded to known files, and cleaned after a thrown callback", () => { + const seen: string[] = []; + for (let i=0; i<2; i++) { + expect(() => withPnpmCommandCwd(["add", "-g", "fixture"], cwd => { + seen.push(cwd); + expect(cwd).not.toBe(tmpdir()); expect(cwd).not.toBe(PNPM_READ_CWD); + expect(readFileSync(cwd + "/pnpm-workspace.yaml", "utf8")).toContain("packages: []"); + if (process.platform !== "win32") expect(lstatSync(cwd).mode & 0o077).toBe(0); + throw new Error("fixture failure"); + })).toThrow("fixture failure"); + } + expect(seen[0]).not.toBe(seen[1]); + for (const cwd of seen) expect(existsSync(cwd)).toBe(false); +}); + +const owner = { commandPath: "/trusted/pnpm", packagePath: "/pkg", globalDir: "/global", globalRoot: "/global", globalBinDir: "/bin" }; +for (const fail of [false,true]) { + test(`actual pnpm spawn options isolate ${fail ? "rollback" : "install"} and registry reads`, () => { + const mutations: string[] = []; + let listCall=0, addCall=0, spawnCalls=0; + const versions = fail ? ["1.0.0","1.0.1","1.0.0"] : ["1.0.0","1.0.1"]; + const result = runPnpmGlobalUpdate({ packageName:"ocx_test", currentVersion:"1.0.0", targetVersion:"1.0.1", tag:"latest", owner, runningPackagePath:"/pkg", + runPnpm:(args:string[],capture=false) => runOwnedPnpm(owner,args,capture,"ignore", ((_bin:unknown,_argv:unknown,options:{cwd:string;env:Record}) => { + spawnCalls++; + expect(options.env.npm_config_ignore_pnpmfile).toBe("true"); + expect(options.env.pnpm_config_ignore_pnpmfile).toBe("true"); + let status=0,stdout=""; + if(args[0]==="add") { + mutations.push(options.cwd); + expect(options.cwd).not.toBe(tmpdir()); expect(options.cwd).not.toBe(PNPM_READ_CWD); + expect(existsSync(options.cwd+"/pnpm-workspace.yaml")).toBe(true); + status=fail && addCall++===0 ? 1 : 0; + } else { + expect(options.cwd).toBe(PNPM_READ_CWD); + if(args[0]==="list") stdout=JSON.stringify([{path:"/global",dependencies:{ocx_test:{version:versions[Math.min(listCall++,versions.length-1)],path:"/pkg"}}}]); + } + return {status,stdout,stderr:"",pid:1,output:[],signal:null}; + }) as never), verify:()=>({ok:true}),verifyShims:()=>({ok:true}) }); + expect(result.ok).toBe(!fail); expect(spawnCalls).toBeGreaterThan(1); + expect(mutations).toHaveLength(fail ? 2 : 1); + for(const cwd of mutations) expect(existsSync(cwd)).toBe(false); + }); +} diff --git a/tests/update/update-refresh.test.ts b/tests/update/update-refresh.test.ts index 9981ac6ff3b..c6ab8ab7ae4 100644 --- a/tests/update/update-refresh.test.ts +++ b/tests/update/update-refresh.test.ts @@ -1,10 +1,13 @@ import { describe, expect, test } from "bun:test"; import { EventEmitter } from "node:events"; +import { dirname } from "node:path"; import { PassThrough } from "node:stream"; +import { fileURLToPath } from "node:url"; import { createRefreshScheduler, RETRY_BASE_MS, STALENESS_TICK_MS, type RefreshDeps } from "../../src/update/refresh-scheduler"; import { latestVersionAsync, pnpmOwner, REGISTRY_DEADLINE_MS, REGISTRY_OUTPUT_LIMIT } from "../../src/update/async-check"; import type { VersionCache } from "../../src/update/notify"; -import type { Channel, Installer } from "../../src/update/index"; +import { checkUpdatePackageIntegrity, latestVersion, resolveCurrentPnpmGlobalOwner, type Channel, type Installer } from "../../src/update/index"; +import { PNPM_READ_CWD, pnpmReadEnvironment } from "../../src/update/pnpm-read-policy.mjs"; function fixture(installer: Installer = "npm", disabled = false, lookupFn?: RefreshDeps["lookup"]) { let now = 1_700_000_000_000; @@ -278,6 +281,99 @@ test("pnpm owner resolution failure is unavailable, not an unowned PATH lookup", expect(spawned).toBe(false); }); +test("pnpm read probes ignore caller project hooks from a trusted directory", () => { + const input = { npm_config_ignore_pnpmfile: "false", NPM_CONFIG_IGNORE_PNPMFILE: "false", SECRET: "retained" }; + expect(pnpmReadEnvironment(input)).toEqual({ npm_config_ignore_pnpmfile: "true", pnpm_config_ignore_pnpmfile: "true", SECRET: "retained" }); + expect(input.npm_config_ignore_pnpmfile).toBe("false"); + expect(PNPM_READ_CWD).toBe(dirname(fileURLToPath(new URL("../../src/update/pnpm-read-policy.mjs", import.meta.url)))); +}); + +test("pnpm registry lookup applies project isolation to its child", async () => { + const child = fakeChild(); + let options: Record | undefined; + const result = latestVersionAsync("latest", "pnpm", { + ownerFn: async () => ({ + commandPath: "/trusted/pnpm", packagePath: "/pkg", globalDir: "/global", + globalRoot: "/global", globalBinDir: "/bin", + }), + spawnFn: ((_bin: string, _args: string[], observed: Record) => { + options = observed; + queueMicrotask(() => { child.stdout.write("2.7.44\n"); child.emit("close", 0); }); + return child; + }) as never, + }); + expect(await result).toBe("2.7.44"); + expect(options?.cwd).toBe(PNPM_READ_CWD); + expect((options?.env as Record).npm_config_ignore_pnpmfile).toBe("true"); +}); + +test("non-pnpm registry lookups keep the caller's working directory", async () => { + const child = fakeChild(); + let options: Record | undefined; + const result = latestVersionAsync("latest", "npm", { + ownerFn: async () => null, + spawnFn: ((_bin: string, _args: string[], observed: Record) => { + options = observed; + queueMicrotask(() => { child.stdout.write("2.7.44\n"); child.emit("close", 0); }); + return child; + }) as never, + }); + expect(await result).toBe("2.7.44"); + expect(options?.cwd).toBeUndefined(); +}); + +const TEST_PNPM_OWNER = { + commandPath: "/trusted/pnpm", packagePath: "/pkg", globalDir: "/global", + globalRoot: "/global", globalBinDir: "/bin", +}; + +test("synchronous pnpm registry lookup applies project isolation", () => { + let options: Record | undefined; + const version = latestVersion("latest", "pnpm", TEST_PNPM_OWNER, (( + _bin: string, + _args: string[], + observed: Record, + ) => { + options = observed; + return { status: 0, stdout: "2.7.44\n", stderr: "", pid: 1, output: [], signal: null }; + }) as never); + expect(version).toBe("2.7.44"); + expect(options?.cwd).toBe(PNPM_READ_CWD); + expect((options?.env as Record).npm_config_ignore_pnpmfile).toBe("true"); +}); + +test("synchronous pnpm integrity probe applies project isolation", () => { + let options: Record | undefined; + const result = checkUpdatePackageIntegrity("2.7.44", (( + _bin: string, + _args: string[], + observed: Record, + ) => { + options = observed; + return { status: 0, stdout: "sha512-AbC123+/=\n", stderr: "", pid: 1, output: [], signal: null }; + }) as never, "pnpm", TEST_PNPM_OWNER); + expect(result.ok).toBe(true); + expect(options?.cwd).toBe(PNPM_READ_CWD); + expect((options?.env as Record).npm_config_ignore_pnpmfile).toBe("true"); +}); + +test("synchronous pnpm owner discovery applies project isolation", () => { + const observed: Array> = []; + const result = resolveCurrentPnpmGlobalOwner("not-a-shim", { + commandPaths: ["/trusted/pnpm"], + spawn: ((_bin: string, _args: string[], options: Record) => { + observed.push(options); + return { status: 1, stdout: "", stderr: "", pid: 1, output: [], signal: null }; + }) as never, + }); + expect(result.ok).toBe(false); + expect(observed.length).toBeGreaterThan(0); + for (const options of observed) { + expect(options.cwd).toBe(PNPM_READ_CWD); + expect((options.env as Record).npm_config_ignore_pnpmfile).toBe("true"); + } +}); + const CAN_RUN_BUN_WORKER = ["darwin", "linux", "win32"].includes(process.platform) && typeof Worker === "function"; From 440e427ae20e49ff653931a62d43d3dfd104bee8 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:49:40 +0900 Subject: [PATCH 27/75] chore(integration): compact the update test layout entries below the size guard Carrying #6048 put scripts/test-layout/layout.json at 2000 lines. Pair seven update-domain explicit entries per line, as 46ee24fa3f did in round 2; every mapping is kept. --- scripts/test-layout/layout.json | 21 +++++++-------------- 1 file changed, 7 insertions(+), 14 deletions(-) diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 331193e92cb..e8c6e95aa0d 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -168,8 +168,7 @@ } }, "explicit": { - "pnpm-command-isolation.test.ts": "update", - "provider-antigravity-quota-retry.test.ts": "providers", + "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", "responses-compaction-recovery-policy.test.ts": "responses", "plugin-loader.test.ts": "lib", "plugin-upstream-hooks.test.ts": "lib", "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", @@ -1810,18 +1809,12 @@ "umans-provider.test.ts": "providers", "uninstall.test.ts": "cli", "update-async-routes.test.ts": "server", - "update-badge.test.ts": "update", - "update-bun-ownership-lease.test.ts": "update", - "update-desktop-badge.test.ts": "update", - "update-desktop-owner.test.ts": "update", - "update-job.test.ts": "update", - "update-worker-launch.test.ts": "update", - "update-notify.test.ts": "update", - "update-npm-cache-preflight.test.ts": "update", - "update-npm-invocation.test.ts": "update", - "update-mise-launcher-target.test.ts": "update", - "update-mise.test.ts": "update", - "update-mise-node-runtime.test.ts": "update", + "update-badge.test.ts": "update", "update-bun-ownership-lease.test.ts": "update", + "update-desktop-badge.test.ts": "update", "update-desktop-owner.test.ts": "update", + "update-job.test.ts": "update", "update-worker-launch.test.ts": "update", + "update-notify.test.ts": "update", "update-npm-cache-preflight.test.ts": "update", + "update-npm-invocation.test.ts": "update", "update-mise-launcher-target.test.ts": "update", + "update-mise.test.ts": "update", "update-mise-node-runtime.test.ts": "update", "update-pnpm.test.ts": "update", "update-refresh.test.ts": "update", "update-stop-classification.test.ts": "update", From c2fa265fb5ec20edf81a906fb5aa0c6293920df4 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 14:50:43 +0900 Subject: [PATCH 28/75] docs(devlog): record train 3 B2 plan, reviews and local proof --- .../_plan/260927_merge_train_3/020_batch2.md | 31 +++++++++++++++++++ 1 file changed, 31 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/020_batch2.md diff --git a/devlog/_plan/260927_merge_train_3/020_batch2.md b/devlog/_plan/260927_merge_train_3/020_batch2.md new file mode 100644 index 00000000000..dab9e137f52 --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/020_batch2.md @@ -0,0 +1,31 @@ +# B2 — non-GUI bug fixes from luvs01 and mdwsk88 + +Base: `dev` `8923ad9835` (after B1 #6059). Branch `codex/train3-b2`. + +| PR | Author | Change | Kimi verdict | Folded | +|---|---|---|---|---| +| #6057 | luvs01 | Compaction-routing fixture drains response state and closes the routing-history index before removing its home (Windows EBUSY) | LAND | — | +| #6035 | luvs01 | A Grok-surface Devin preflight 429 binds the usage it reports | LAND | — | +| #6022 | mdwsk88 | CodeBuddy capture refuses partial same-index blocks, cross-index stop/delta, and enforces the 16-call limit at open | LAND | — | +| #6047 | luvs01 | Sidecar probe leases are released on pre-dispatch rejections and when streamed sidecar responses settle | LAND (`core.ts` 209 of 210) | — | +| #6046 | luvs01 | Turning CLI first-party off pins Desktop's mode and reports `shared_proxy_retained` | LAND-WITH-FIXES | `a3d40e1f63`: the human CLI output now prints the warning and how to release the env | +| #6036 | luvs01 | Service ownership compares recorded homes by physical directory | LAND-WITH-FIXES | `43dbcee265`: a missing, differently spelled recorded home stays foreign (ENOENT/ENOTDIR), matching its own docs and the WP13 restore contract that its head failed in CI | +| #6038 | luvs01 | Devin web search spends the routed provider's OAuth account and tenant | LAND; security review: no blocker | — | +| #6048 | luvs01 | pnpm update probes and mutations run isolated from the caller's project | LAND | `440e427ae2`: layout entries paired to stay under the size guard | + +Not folded: #6046's GUI notice (out of scope for a non-GUI round) and its idempotent-PUT nit. + +## Aside evidence + +Captures in `.tmp/aside/pull-.txt`. None of the eight links an issue. #6036 still shows Ingwannu's +CHANGES_REQUESTED review from an earlier head; Kimi confirmed the requested `unknown` verdict is in the head, and +`43dbcee265` fixes the CI regression that remained. #6038 shows two approvals. #6057 and #6022 carry draft history. + +## Local proof at `440e427ae2` + +- `bun run typecheck`, `bun run structure:check`, `bun run privacy:scan`: exit 0. +- 13 focused files: 428 pass, 1 skip, 10 fail; the 10 are `api-key-scope-alpha-search` cases that trip this + worktree's protected-home guard (it lives under `~/.codex`). In a `/tmp` worktree at the same head, that file and + `service-sqlite-home` pass 15/15. +- `tests/service/` in `/tmp`: 709 pass, 3 `shutdown-launcher` failures that need a free proxy port on this machine. +- File-size ratchet and both test-layout guards: 27 pass. From c609894708bbd1d85c632741720b25f74a35009f Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 15:11:51 +0900 Subject: [PATCH 29/75] docs(devlog): plan train 3 B3 --- .../_plan/260927_merge_train_3/030_batch3.md | 21 +++++++++++++++++++ 1 file changed, 21 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/030_batch3.md diff --git a/devlog/_plan/260927_merge_train_3/030_batch3.md b/devlog/_plan/260927_merge_train_3/030_batch3.md new file mode 100644 index 00000000000..ee95ac01e0e --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/030_batch3.md @@ -0,0 +1,21 @@ +# B3 — quota activation, update launcher, link join, settings reads, account selection, combo exhaustion + +Base: `dev` `06d7914e6a` (after B2 #6061). Branch `codex/train3-b3`. + +Previous D (B1+B2): both batches landed with exact-head CI (35 success, 5 path-skipped each). Direction kept: serialized +carries with review fixes as separate commits. Lesson: build only inside B, after A. + +| Item | Author | Plan | Fixes to fold | +|---|---|---|---| +| #6049 | luvs01 | Carry. Bounded, non-blocking read of the global Codex config on the settings poll path. | Layout registries: union with #6048's compaction. | +| #6037 | luvs01 | Carry. `systemd-run` resolves only from trusted root-owned paths, probed off the event loop. | `src/update/job.ts` import conflict: keep both imports (file lands at 1999 of 2000 lines). | +| #6042 | luvs01 | Carry. The Remote Link join key leaves only after the tunnel's listener ownership is proven twice. | None required; the connect-phase race stays documented in `structure/remote-link.md` as the PR states. | +| #6020 | terrytan95 | Carry; resolves #6018. Deadline-first quota activation with bounded backoff. | Retry records carry the credential generation, so an old credential's failure cannot hold back a replacement; a local `native main busy` refusal retries in one minute without growing the backoff; drop the duplicate delete. | +| #6056 | luvs01 | Close as superseded by #6020. On dev the retained earliest deadline already starts an idle window once, and #6020 stops the polling. | — | +| #6050 | luvs01 | Carry. `ocx account clear`; an account id `auto` wins over the reserved word; clearing works while main is paused. | Rewrite the dev test that pinned the old 409; revert its unrelated `shadow` default and `strategy` doc hunks; union the layout registries. | +| #5494 | (issue, found through Aside) | Implement. A 429 whose body says the token-plan quota "has been exhausted" is account exhaustion, so the combo target takes the long hold instead of being offered again every 60 s. | Regression test next to the combo exhaustion tests. | + +Held: #6027 (owner's three blockers are still open on a draft head), #6030 and #6003 (drafts), GUI PRs. + +Security-boundary items: #6037 (updater command execution) and #6042 (link join credential) have dedicated Kimi +security reviews with no blocker recorded in this unit. From e964e04fa17c2b29987b2f4134217ee75c84d475 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 15:27:26 +0900 Subject: [PATCH 30/75] docs(devlog): fold the B3 audit into the plan --- devlog/_plan/260927_merge_train_3/030_batch3.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/030_batch3.md b/devlog/_plan/260927_merge_train_3/030_batch3.md index ee95ac01e0e..776d3526e5b 100644 --- a/devlog/_plan/260927_merge_train_3/030_batch3.md +++ b/devlog/_plan/260927_merge_train_3/030_batch3.md @@ -19,3 +19,17 @@ Held: #6027 (owner's three blockers are still open on a draft head), #6030 and # Security-boundary items: #6037 (updater command execution) and #6042 (link join credential) have dedicated Kimi security reviews with no blocker recorded in this unit. + +## Audit (Kimi, NEAR-PASS) and folded decisions + +- #6020 busy path: a named `NativeMainBusyError`; the retry record keeps its prior `delay` and sets `after = now + 60 s`. +- #6020 `main account unavailable`: stays in the growing backoff. It is keyed by generation, so a token that + arrives later starts clean. +- #6020 both retry maps (`retryAfterByAccount`, `quotaRefreshAfterByAccount`) carry the credential generation; a + record from another generation is dropped when read. The generation is captured before `warm()`/`refresh()` and a + failure is not recorded when it changed during the await. The second same-tick `hasScheduledWindows` delete goes, + because the generation check covers it and a leftover metadata backoff cannot gate a scheduled account. +- #6020 tests: replacement during the await, repeated busy refusal then release, reauth then rotation. +- #5494: the regex is anchored to the token-plan phrasing, + `/usage limit (?:has been )?reached|token-plan\s+\S+\s+quota has been exhausted/`, with a negative case for + "quota exhausted for this minute". The hold is the existing ten-minute exhaustion cap, not the announced reset. From c79fe409c6294a86feeab3bd98c504b26b0cca0a Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 15:27:46 +0900 Subject: [PATCH 31/75] fix(settings): bound Codex ownership config reads (#6049) Carried from #6049 into merge train round 3. Co-authored-by: Epinephrine --- scripts/test-layout/layout.json | 2 +- src/codex/desktop-switches.ts | 12 +- src/codex/inject/bounded-config-reader.ts | 69 ++++++ src/codex/inject/config-toml.ts | 18 ++ src/codex/project-config-warnings.ts | 93 +++++++-- structure/config.md | 12 +- .../project-config-warning-snapshot.test.ts | 42 ++++ .../project-config-warnings.test.ts | 21 ++ .../settings-desktop-switch-apply.test.ts | 196 +++++++++++++++++- tests/fixtures/test-layout-expected.json | 1 + 10 files changed, 432 insertions(+), 34 deletions(-) create mode 100644 src/codex/inject/bounded-config-reader.ts create mode 100644 tests/codex-integration/project-config-warning-snapshot.test.ts diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index e8c6e95aa0d..b2ecb246232 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -168,7 +168,7 @@ } }, "explicit": { - "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", + "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "project-config-warning-snapshot.test.ts": "codex-integration", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", "responses-compaction-recovery-policy.test.ts": "responses", "plugin-loader.test.ts": "lib", "plugin-upstream-hooks.test.ts": "lib", "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", diff --git a/src/codex/desktop-switches.ts b/src/codex/desktop-switches.ts index 3bd7384b156..a4b99a85fd8 100644 --- a/src/codex/desktop-switches.ts +++ b/src/codex/desktop-switches.ts @@ -116,14 +116,16 @@ export function describeCodexDesktopSwitches( */ export async function observedCodexDesktopSwitchApply(): Promise { // Same lazy boundary as applyCodexConfigInjection: the ownership predicate lives in the - // injection graph, which the settings read path must not pull in at module scope. - const { currentExternalCodexModelProvider } = await import("./inject/config-toml"); + // injection graph, which the settings read path must not pull in at module scope. This + // path uses the bounded variant — a special or oversized config.toml must answer + // "undetermined", never stall a settings read the way an unbounded readFileSync would. + const { observedExternalCodexModelProvider } = await import("./inject/config-toml"); let provider: string | null; try { - provider = currentExternalCodexModelProvider(); + provider = observedExternalCodexModelProvider(); } catch (error) { - // A present-but-unreadable config.toml (permissions, deletion racing existsSync) - // must not take down the whole settings report. The undetermined reason keeps the + // A present-but-unreadable config.toml (permissions, deletion racing the bounded + // read) must not take down the whole settings report. The undetermined reason keeps the // reporting contract honest: effective values and the sign-in answer stay null instead // of presenting local state a foreign provider may still control. return { diff --git a/src/codex/inject/bounded-config-reader.ts b/src/codex/inject/bounded-config-reader.ts new file mode 100644 index 00000000000..78a7ce95030 --- /dev/null +++ b/src/codex/inject/bounded-config-reader.ts @@ -0,0 +1,69 @@ +import { closeSync, constants, fstatSync, openSync, readSync, statSync, type Stats } from "node:fs"; + +const MAX_CODEX_CONFIG_BYTES = 1024 * 1024; + +/** + * Read config.toml the way Codex and the injector resolve it — a symlink's target IS the + * config — without blocking on a special file or buffering without bound. + * + * `null` means only "absent at the initial lookup". A path that vanishes or is swapped + * underneath the read throws the changed-file error instead, because the observation is + * then undetermined rather than negative: the caller must not report a config the probe + * watched disappear as simply not there. + */ +export function readBoundedCodexConfig(path: string): string | null { + let fd: number | undefined; + try { + let namedBefore: Stats; + try { + namedBefore = statSync(path); + } catch (error) { + const code = (error as NodeJS.ErrnoException).code; + if (code === "ENOENT" || code === "ENOTDIR") return null; + throw error; + } + if (!namedBefore.isFile() || namedBefore.size > MAX_CODEX_CONFIG_BYTES) { + throw new Error("config.toml is not a bounded regular file"); + } + // Deliberately no O_NOFOLLOW: Codex and the injector read through a symlinked + // config.toml, so refusing the link here would disagree with the writes this probe + // stands in front of. O_NONBLOCK is what keeps a FIFO — linked or direct — from + // stalling the open; the descriptor checks below still reject anything non-regular. + const guardedFlags = process.platform === "win32" + ? 0 + : (constants.O_NONBLOCK ?? 0); + fd = openSync(path, constants.O_RDONLY | guardedFlags); + const before = fstatSync(fd); + if (before.dev !== namedBefore.dev || before.ino !== namedBefore.ino) { + throw new Error("config.toml changed while it was read"); + } + if (!before.isFile() || before.size > MAX_CODEX_CONFIG_BYTES) { + throw new Error("config.toml is not a bounded regular file"); + } + + const buffer = Buffer.allocUnsafe(before.size + 1); + let bytesRead = 0; + while (bytesRead < buffer.length) { + const count = readSync(fd, buffer, bytesRead, buffer.length - bytesRead, null); + if (count === 0) break; + bytesRead += count; + } + const after = fstatSync(fd); + const namedAfter = statSync(path); + if (bytesRead !== before.size || after.size !== before.size + || after.mtimeMs !== before.mtimeMs || after.ctimeMs !== before.ctimeMs + || !namedAfter.isFile() + || namedAfter.dev !== before.dev || namedAfter.ino !== before.ino) { + throw new Error("config.toml changed while it was read"); + } + return buffer.toString("utf8", 0, bytesRead); + } catch (error) { + const code = (error as NodeJS.ErrnoException).code; + if (code === "ENOENT" || code === "ENOTDIR") { + throw new Error("config.toml changed while it was read"); + } + throw error; + } finally { + if (fd !== undefined) closeSync(fd); + } +} diff --git a/src/codex/inject/config-toml.ts b/src/codex/inject/config-toml.ts index e91c055bb26..761c09549dd 100644 --- a/src/codex/inject/config-toml.ts +++ b/src/codex/inject/config-toml.ts @@ -18,6 +18,7 @@ import { resolveCodexConfigPath, tomlString, } from "../paths"; +import { readBoundedCodexConfig } from "./bounded-config-reader"; import { type CodexRoutingTarget, providerBaseHost, @@ -33,11 +34,28 @@ export function externalCodexModelProvider(content: string): string | null { : null; } +/** + * The ownership answer for read/write paths — inject, sync, connect, restore, and the + * shutdown gate. It deliberately reads the whole file like Codex does (links included): + * a large or link-mediated config is still a valid config, and these callers must + * classify it exactly rather than degrade to "undetermined". + */ export function currentExternalCodexModelProvider(): string | null { if (!existsSync(CODEX_CONFIG_PATH)) return null; return externalCodexModelProvider(readFileSync(CODEX_CONFIG_PATH, "utf8")); } +/** + * The same ownership answer for read-only observation (the settings GET / poll path), + * through a bounded read so a special or oversized config.toml cannot stall a request. + * A present-but-unreadable config throws so the caller reports undetermined ownership + * instead of "none". + */ +export function observedExternalCodexModelProvider(): string | null { + const content = readBoundedCodexConfig(CODEX_CONFIG_PATH); + return content === null ? null : externalCodexModelProvider(content); +} + /** * Detect the file's dominant line ending. Every transform in this module is LF-pure * (split("\n") + hard "\n" joins), so CRLF configs (Windows-edited config.toml) are diff --git a/src/codex/project-config-warnings.ts b/src/codex/project-config-warnings.ts index 41a11f78649..183f76c4e83 100644 --- a/src/codex/project-config-warnings.ts +++ b/src/codex/project-config-warnings.ts @@ -5,13 +5,13 @@ import { fstatSync, lstatSync, openSync, - readFileSync, readSync, realpathSync, } from "node:fs"; import path, { dirname, join, resolve } from "node:path"; import { expandUserPath } from "../config"; import { defaultCodexHome } from "./home"; +import { readBoundedCodexConfig } from "./inject/bounded-config-reader"; import { readRootTomlString } from "./paths"; import { truncateRetainedUtf8 } from "../lib/admission"; @@ -53,7 +53,8 @@ function resolveCodexConfigPath(): string { return join(home, "config.toml"); } -export type ProjectCodexConfigIssueCode = "model_providers_table" | "profile_selector" | "model_provider_root"; +export type ProjectCodexConfigIssueCode = "model_providers_table" | "profile_selector" | "model_provider_root" + | "global_config_unreadable"; export interface ProjectCodexConfigWarning { path: string; @@ -265,17 +266,18 @@ export function resolveEffectiveProjectModelProvider(content: string): Effective /** True when global Codex config routes through the opencodex proxy. */ export function isGlobalOpencodexRoutingActive( codexConfigPath: string = resolveCodexConfigPath(), - content?: string, + content?: string | null, ): boolean { let text = content; if (text === undefined) { - if (!existsSync(codexConfigPath)) return false; try { - text = readFileSync(codexConfigPath, "utf-8"); + text = readBoundedCodexConfig(codexConfigPath) ?? undefined; } catch { return false; } + if (text === undefined) return false; } + if (text === null) return false; if (hasInjectedOpenaiBaseUrl(text)) return true; if (readRootTomlString(text, "model_provider") === "opencodex") return true; return false; @@ -379,6 +381,8 @@ export function discoverProjectCodexConfigPaths(options: { cwd?: string; codexConfigPath?: string; maxWalkParents?: number; + /** Explicit null keeps an absent/unreadable observation; undefined permits a fresh read. */ + globalContent?: string | null; } = {}): string[] { const found = new Set(); const codexConfigPath = options.codexConfigPath ?? resolveCodexConfigPath(); @@ -416,15 +420,16 @@ export function discoverProjectCodexConfigPaths(options: { cwd = parent; } - if (existsSync(codexConfigPath)) { - try { - const global = readFileSync(codexConfigPath, "utf-8"); + try { + const global = options.globalContent === undefined + ? readBoundedCodexConfig(codexConfigPath) : options.globalContent; + if (global !== null) { for (const projectPath of parseTrustedProjectPathsFromCodexConfig(global)) { addIfExists(projectPath); } - } catch { - /* ignore unreadable global config */ } + } catch { + /* ignore unreadable global config */ } return [...found]; @@ -437,10 +442,32 @@ export function collectProjectCodexConfigWarnings(options: { } = {}): ProjectCodexConfigWarning[] { const codexConfigPath = options.codexConfigPath ?? resolveCodexConfigPath(); const requireRouting = options.requireOpencodexRouting ?? true; - if (requireRouting && !isGlobalOpencodexRoutingActive(codexConfigPath)) return []; + + // The routing question has three answers: active, inactive, and unreadable. An oversized + // or swapped-underneath global config must not silently collapse to "inactive" — that + // would erase both project-bypass coverage and trusted-path discovery without a trace. + let globalContent: string | null = null; + let globalUnreadable = false; + try { + globalContent = readBoundedCodexConfig(codexConfigPath); + } catch { + globalUnreadable = true; + } + if (requireRouting && !globalUnreadable + && !isGlobalOpencodexRoutingActive(codexConfigPath, globalContent)) { + return []; + } const warnings: ProjectCodexConfigWarning[] = []; - for (const path of discoverProjectCodexConfigPaths({ cwd: options.cwd, codexConfigPath })) { + if (globalUnreadable) { + warnings.push({ + path: codexConfigPath, + code: "global_config_unreadable", + detail: "unreadable", + message: "The global Codex config could not be read within the 1 MiB bound — whether it routes through OpenCodex, and which projects it declares trusted, is undetermined.", + }); + } + for (const path of discoverProjectCodexConfigPaths({ cwd: options.cwd, codexConfigPath, globalContent })) { const content = readBoundedProjectConfig(path); if (content !== null) warnings.push(...analyzeProjectCodexConfig(content, path)); } @@ -480,6 +507,8 @@ export function summarizeProjectCodexIssue(warning: ProjectCodexConfigWarning): return warning.profileName ? `profile="${warning.profileName}"` : `model_provider="${warning.detail}"`; case "model_provider_root": return `model_provider="${warning.detail}"`; + case "global_config_unreadable": + return "config.toml unreadable or oversized"; } } @@ -501,6 +530,8 @@ export interface ProjectCodexConfigWarningGroup { path: string; issues: string[]; bypass: string; + /** True when the group is the global-config-unreadable caveat, not a project bypass. */ + globalUnreadable?: boolean; } export function groupProjectCodexConfigWarningsByPath( @@ -512,22 +543,34 @@ export function groupProjectCodexConfigWarningsByPath( list.push(warning); grouped.set(warning.path, list); } - return [...grouped.entries()].map(([path, pathWarnings]) => ({ - path, - issues: pathWarnings.map(summarizeProjectCodexIssue), - bypass: explainProjectConfigBypass(pathWarnings), - })); + return [...grouped.entries()].map(([path, pathWarnings]) => { + const globalUnreadable = pathWarnings.every(warning => warning.code === "global_config_unreadable"); + return { + path, + issues: pathWarnings.map(summarizeProjectCodexIssue), + bypass: globalUnreadable ? pathWarnings[0]!.message : explainProjectConfigBypass(pathWarnings), + ...(globalUnreadable ? { globalUnreadable } : {}), + }; + }); } export function formatProjectCodexConfigWarningsForDoctor(warnings: ProjectCodexConfigWarning[]): string[] { const grouped = groupProjectCodexConfigWarningsByPath(warnings); if (grouped.length === 0) return []; const lines: string[] = []; - for (const { path, issues, bypass } of grouped) { + let hasBypassEntries = false; + for (const { path, issues, bypass, globalUnreadable } of grouped) { lines.push(` -- ${relPath(path)} — ${issues.join(", ")}`); lines.push(` ${bypass}`); + if (globalUnreadable) { + lines.push(" fix: keep the global config.toml a readable regular file within the 1 MiB bound"); + } else { + hasBypassEntries = true; + } + } + if (hasBypassEntries) { + lines.push(" fix: remove those entries so OpenCodex proxy routing applies in this project"); } - lines.push(" fix: remove those entries so OpenCodex proxy routing applies in this project"); return lines; } @@ -535,11 +578,19 @@ export function formatProjectCodexConfigWarningsForConsole(warnings: ProjectCode const grouped = groupProjectCodexConfigWarningsByPath(warnings); if (grouped.length === 0) return []; const lines = ["⚠️ Project Codex config bypasses OpenCodex:"]; - for (const { path, issues, bypass } of grouped) { + let hasBypassEntries = false; + for (const { path, issues, bypass, globalUnreadable } of grouped) { lines.push(` ${relPath(path)} — ${issues.join(", ")}`); lines.push(` ${bypass}`); + if (globalUnreadable) { + lines.push(" fix: keep the global config.toml a readable regular file within the 1 MiB bound"); + } else { + hasBypassEntries = true; + } + } + if (hasBypassEntries) { + lines.push(" fix: remove those entries so OpenCodex proxy routing applies in this project"); } - lines.push(" fix: remove those entries so OpenCodex proxy routing applies in this project"); return lines; } diff --git a/structure/config.md b/structure/config.md index f6acb718130..1207476bc33 100644 --- a/structure/config.md +++ b/structure/config.md @@ -203,12 +203,12 @@ provider with no name and rejects the whole config rather than one thread, which than the branding it would remove — so a blank, over-length, or control-character value falls back to the default instead of being written. -Read-only doctor and project-routing diagnostics use a lightweight root/table TOML reader rather -than mutating or normalizing the user's file. That reader must lexically skip both basic and literal -multiline string bodies: instruction prose can contain key-shaped examples and `[table]` snippets, -which are data rather than configuration. Diagnostic result objects may retain the real path for -local correlation, but every formatted doctor line must pass it through the shared user-path -redaction boundary before display. +Read-only global ownership/doctor diagnostics follow links only to bounded regular files; an absent +lookup reads as none and an unreadable/changed observation reports undetermined ownership. +Project discovery instead skips links/oversized entries, and its guarded reader skips unsafe files. +Each project-warning collection shares one global snapshot for routing and trusted-path discovery, +including explicit absence or read failure. TOML parsing skips multiline string bodies rather than +reading prose as configuration; formatted doctor paths pass through user-path redaction. > Decision record: [ADR-0017](decisions/ADR-0017-config-injection.md) diff --git a/tests/codex-integration/project-config-warning-snapshot.test.ts b/tests/codex-integration/project-config-warning-snapshot.test.ts new file mode 100644 index 00000000000..e3e1996902e --- /dev/null +++ b/tests/codex-integration/project-config-warning-snapshot.test.ts @@ -0,0 +1,42 @@ +import { expect, spyOn, test } from "bun:test"; +import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import * as bounded from "../../src/codex/inject/bounded-config-reader"; +import { collectProjectCodexConfigWarnings, discoverProjectCodexConfigPaths, isGlobalOpencodexRoutingActive } from "../../src/codex/project-config-warnings"; +import { removeTreeWithRetry } from "../helpers/remove-tree"; + +for (const kind of ["present", "absent", "unreadable"] as const) { + test(`project warnings read exactly one global ${kind} snapshot`, () => { + const root = mkdtempSync(join(tmpdir(), "ocx-warning-snapshot-")); + const global = join(root, "global.toml"); + const project = join(root, "project"); + mkdirSync(join(project, ".codex"), { recursive: true }); + const projectConfig = join(project, ".codex", "config.toml"); + writeFileSync(projectConfig, 'model_provider = "external"\n'); + const text = 'model_provider = "opencodex"\n'; + writeFileSync(global, text); + const read = spyOn(bounded, "readBoundedCodexConfig").mockImplementation(() => { + if (kind === "unreadable") throw new Error("fixture read failure"); + return kind === "absent" ? null : text; + }); + try { + const warnings = collectProjectCodexConfigWarnings({ cwd: project, codexConfigPath: global }); + expect(read).toHaveBeenCalledTimes(1); + expect(warnings.some(w => w.code === "global_config_unreadable")).toBe(kind === "unreadable"); + expect(warnings.some(w => w.path === projectConfig)).toBe(kind !== "absent"); + } finally { read.mockRestore(); removeTreeWithRetry(root); } + }); +} + +test("explicit absent snapshots never become fresh global reads", () => { + const root = mkdtempSync(join(tmpdir(), "ocx-warning-absent-")); + const global = join(root, "global.toml"); + writeFileSync(global, 'model_provider = "opencodex"\n'); + const read = spyOn(bounded, "readBoundedCodexConfig").mockImplementation(() => { throw new Error("unexpected reread"); }); + try { + expect(isGlobalOpencodexRoutingActive(global, null)).toBe(false); + expect(discoverProjectCodexConfigPaths({ cwd: root, codexConfigPath: global, maxWalkParents: 1, globalContent: null })).toEqual([]); + expect(read).not.toHaveBeenCalled(); + } finally { read.mockRestore(); removeTreeWithRetry(root); } +}); diff --git a/tests/codex-integration/project-config-warnings.test.ts b/tests/codex-integration/project-config-warnings.test.ts index f130522c684..c4ebe5cff97 100644 --- a/tests/codex-integration/project-config-warnings.test.ts +++ b/tests/codex-integration/project-config-warnings.test.ts @@ -393,6 +393,27 @@ name = "anthropic" .filter(warning => warning.path === projectConfigPath); expect(second.length).toBe(0); }); + + test("an oversized global config is reported as unreadable instead of silently inactive", () => { + const codexConfigPath = join(process.env.CODEX_HOME!, "config.toml"); + mkdirSync(process.env.CODEX_HOME!, { recursive: true }); + writeFileSync(codexConfigPath, `# ${"x".repeat(1024 * 1024)}`); + const warnings = collectProjectCodexConfigWarnings({ cwd: testDir, codexConfigPath }); + const global = warnings.find(warning => warning.code === "global_config_unreadable"); + expect(global?.path).toBe(codexConfigPath); + }); + + test("an oversized global config still surfaces bypasses found by walking parents", () => { + const codexConfigPath = join(process.env.CODEX_HOME!, "config.toml"); + mkdirSync(process.env.CODEX_HOME!, { recursive: true }); + writeFileSync(codexConfigPath, `# ${"x".repeat(1024 * 1024)}`); + const projectConfigPath = join(testDir, ".codex", "config.toml"); + mkdirSync(join(testDir, ".codex"), { recursive: true }); + writeFileSync(projectConfigPath, `model_provider = "anthropic"`); + const warnings = collectProjectCodexConfigWarnings({ cwd: testDir, codexConfigPath }); + expect(warnings.some(warning => warning.code === "global_config_unreadable")).toBe(true); + expect(warnings.some(warning => warning.path === projectConfigPath)).toBe(true); + }); }); describe("explainProjectConfigBypass", () => { diff --git a/tests/config/settings-desktop-switch-apply.test.ts b/tests/config/settings-desktop-switch-apply.test.ts index a72a28cd184..297189fffe6 100644 --- a/tests/config/settings-desktop-switch-apply.test.ts +++ b/tests/config/settings-desktop-switch-apply.test.ts @@ -1,8 +1,9 @@ import { expect, spyOn, test } from "bun:test"; import { spawnSync } from "node:child_process"; -import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs"; +import { mkdirSync, mkdtempSync, symlinkSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { readBoundedCodexConfig } from "../../src/codex/inject/bounded-config-reader"; import { removeTreeWithRetry } from "../helpers/remove-tree"; import { repoRoot } from "../helpers/repo-root"; @@ -59,6 +60,35 @@ function runIsolatedSettingsRequest(options: { expect(line).toBeDefined(); return JSON.parse(line!) as { status: number; body: Record }; } + +/** + * Same isolation boundary as the settings cases, for the restore path: CODEX_HOME must + * be fixed before the module graph binds CODEX_CONFIG_PATH. The child's last stdout line + * is the JSON result; earlier lines may be the restore machinery's own logs. + */ +function runIsolatedCodexScript(options: { + root: string; + codexHome: string; + script: string; +}): Record { + const child = spawnSync(process.execPath, ["--eval", options.script], { + cwd: repoRoot(), + env: { + ...process.env, + CODEX_HOME: options.codexHome, + OPENCODEX_HOME: join(options.root, "opencodex"), + }, + encoding: "utf8", + timeout: 10_000, + }); + if (child.status !== 0) { + const cause = child.error ? ` (${child.error.name}: ${child.error.message})` : ""; + throw new Error(`isolated codex script failed (status=${child.status} signal=${child.signal})${cause}: ${child.stderr || child.stdout}`); + } + const line = child.stdout.trim().split("\n").filter(Boolean).at(-1); + expect(line).toBeDefined(); + return JSON.parse(line!) as Record; +} test("PUT /api/settings reports Codex write-lock contention as retryable", async () => { const root = mkdtempSync(join(tmpdir(), "ocx-settings-desktop-switch-")); const codexHome = join(root, "codex"); @@ -214,6 +244,42 @@ test("GET /api/settings survives an unreadable config.toml during ownership dete } }, 15_000); +test.skipIf(process.platform === "win32")( + "GET /api/settings refuses a config.toml FIFO without blocking", + () => { + const root = mkdtempSync(join(tmpdir(), "ocx-settings-fifo-cfg-")); + const codexHome = join(root, "codex"); + mkdirSync(codexHome, { recursive: true }); + const fifo = spawnSync("mkfifo", [join(codexHome, "config.toml")], { encoding: "utf8" }); + expect(fifo.status).toBe(0); + + try { + const response = runIsolatedSettingsRequest({ + root, + codexHome, + routeConfig: ISOLATED_PROVIDER_CONFIG, + scriptBody: ` + const request = new Request("http://127.0.0.1:10100/api/settings", { + headers: { host: "127.0.0.1:10100" }, + }); + const response = await handleManagementAPI(request, new URL(request.url), config, { + getCachedStartupHealth: async () => startupHealthFixture(), + }); + `, + }); + expect(response.status).toBe(200); + expect(response.body).toMatchObject({ + codexDesktopSwitches: { + apply: { applied: false, reason: "ownership_undetermined", retryable: true }, + }, + }); + } finally { + removeTreeWithRetry(root); + } + }, + 15_000, +); + test("PUT /api/settings keeps the undetermined-ownership explanation on a locked save", () => { // clientIntegrations.codex = false trips the apply gate before the injector runs, and an // unreadable config.toml leaves ownership undetermined. The locked save must still report @@ -262,6 +328,134 @@ test("PUT /api/settings keeps the undetermined-ownership explanation on a locked } }, 15_000); +test("readBoundedCodexConfig returns null only for a config absent at lookup", () => { + const root = mkdtempSync(join(tmpdir(), "ocx-bounded-reader-")); + try { + expect(readBoundedCodexConfig(join(root, "config.toml"))).toBeNull(); + writeFileSync(join(root, "config.toml"), 'model_provider = "custom"\n'); + expect(readBoundedCodexConfig(join(root, "config.toml"))).toContain('"custom"'); + // Present but unreadable-as-a-bounded-regular-file is a throw, not a null. + mkdirSync(join(root, "as-dir.toml")); + expect(() => readBoundedCodexConfig(join(root, "as-dir.toml"))).toThrow(); + writeFileSync(join(root, "big.toml"), `# ${"x".repeat(1024 * 1024)}\nmodel = "gpt-5.5"\n`); + expect(() => readBoundedCodexConfig(join(root, "big.toml"))).toThrow(); + } finally { + removeTreeWithRetry(root); + } +}); + +test.skipIf(process.platform === "win32")( + "readBoundedCodexConfig resolves a symlinked config to a bounded regular target", + () => { + const root = mkdtempSync(join(tmpdir(), "ocx-bounded-link-")); + try { + writeFileSync(join(root, "dotfiles-codex.toml"), 'model_provider = "custom"\n'); + symlinkSync(join(root, "dotfiles-codex.toml"), join(root, "config.toml")); + expect(readBoundedCodexConfig(join(root, "config.toml"))).toContain('"custom"'); + // A link does not launder an unsafe target: the descriptor check still refuses it. + symlinkSync("/dev/null", join(root, "null.toml")); + expect(() => readBoundedCodexConfig(join(root, "null.toml"))).toThrow(); + symlinkSync(join(root, "missing.toml"), join(root, "dangling.toml")); + expect(readBoundedCodexConfig(join(root, "dangling.toml"))).toBeNull(); + } finally { + removeTreeWithRetry(root); + } + }, +); + +test.skipIf(process.platform === "win32")( + "GET /api/settings reads ownership through a symlinked config.toml", + () => { + // Codex and the injector read the link's target, so the bounded observation must + // too — otherwise settings reports undetermined for a config that plainly selects + // an external provider. + const root = mkdtempSync(join(tmpdir(), "ocx-settings-link-cfg-")); + const codexHome = join(root, "codex"); + mkdirSync(codexHome, { recursive: true }); + writeFileSync(join(root, "dotfiles-codex.toml"), 'model_provider = "custom"\n'); + symlinkSync(join(root, "dotfiles-codex.toml"), join(codexHome, "config.toml")); + + try { + const response = runIsolatedSettingsRequest({ + root, + codexHome, + routeConfig: ISOLATED_PROVIDER_CONFIG, + scriptBody: ` + const request = new Request("http://127.0.0.1:10100/api/settings", { + headers: { host: "127.0.0.1:10100" }, + }); + const response = await handleManagementAPI(request, new URL(request.url), config, { + getCachedStartupHealth: async () => startupHealthFixture(), + }); + `, + }); + expect(response.status).toBe(200); + expect(response.body).toMatchObject({ + codexDesktopSwitches: { + apply: { applied: false, reason: "external_provider", retryable: false }, + }, + }); + } finally { + removeTreeWithRetry(root); + } + }, + 15_000, +); + +test.skipIf(process.platform === "win32")( + "native restore still classifies a symlinked config.toml's target", + () => { + // Regression for the shared ownership probe: a link to a small regular config must + // produce the external-provider result, not an early exit that leaves injected + // routing pointed at a stopped proxy. + const root = mkdtempSync(join(tmpdir(), "ocx-restore-link-cfg-")); + const codexHome = join(root, "codex"); + mkdirSync(codexHome, { recursive: true }); + writeFileSync(join(root, "dotfiles-codex.toml"), 'model_provider = "custom"\n'); + symlinkSync(join(root, "dotfiles-codex.toml"), join(codexHome, "config.toml")); + + try { + const result = runIsolatedCodexScript({ + root, + codexHome, + script: ` + const { restoreNativeCodex } = await import("./src/codex/inject"); + const result = restoreNativeCodex(); + console.log(JSON.stringify({ success: result.success, externalProvider: result.externalProvider ?? null })); + `, + }); + expect(result).toMatchObject({ success: true, externalProvider: "custom" }); + } finally { + removeTreeWithRetry(root); + } + }, + 15_000, +); + +test("native restore tolerates a config.toml over the observation bound", () => { + // A valid config larger than the 1 MiB observation bound must still classify as + // external — the read/write ownership probe is not the bounded settings reader. + const root = mkdtempSync(join(tmpdir(), "ocx-restore-big-cfg-")); + const codexHome = join(root, "codex"); + mkdirSync(codexHome, { recursive: true }); + writeFileSync(join(codexHome, "config.toml"), `# ${"x".repeat(1024 * 1024)}\nmodel_provider = "custom"\n`); + + try { + const result = runIsolatedCodexScript({ + root, + codexHome, + script: ` + const { restoreNativeCodex } = await import("./src/codex/inject"); + const result = restoreNativeCodex(); + console.log(JSON.stringify({ success: result.success, externalProvider: result.externalProvider ?? null })); + `, + }); + expect(result).toMatchObject({ success: true, externalProvider: "custom" }); + } finally { + removeTreeWithRetry(root); + } +}, 15_000); + test("PUT /api/settings reports external Codex ownership when the integration is disabled", () => { // clientIntegrations.codex = false trips the apply gate before the injector runs, so // the ownership classification has to happen inside applyCodexConfigInjection itself. diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 903440aa89d..97b9abfedc3 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1,5 +1,6 @@ { "pnpm-command-isolation.test.ts": "update", + "project-config-warning-snapshot.test.ts": "codex-integration", "provider-antigravity-quota-retry.test.ts": "providers", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", From 33b1920cce7d01253a40fb53c14c238ad400a8c1 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 15:27:49 +0900 Subject: [PATCH 32/75] fix(link): gate join credential on tunnel survival (#6042) Carried from #6042 into merge train round 3. Co-authored-by: Epinephrine --- src/client/link-join.ts | 91 +++++++-- src/server/port-reclaim.ts | 202 ++++++++++++++++---- structure/remote-link.md | 6 +- tests/server/link-join-route.test.ts | 264 +++++++++++++++++++++++++-- tests/server/port-reclaim.test.ts | 169 +++++++++++++++++ 5 files changed, 666 insertions(+), 66 deletions(-) diff --git a/src/client/link-join.ts b/src/client/link-join.ts index 5059a4a5515..ad2fe285de9 100644 --- a/src/client/link-join.ts +++ b/src/client/link-join.ts @@ -1,6 +1,7 @@ import { randomBytes } from "node:crypto"; import { hostname } from "node:os"; import { isPortAvailable } from "../server/ports"; +import { scanListenPidsForAddress, type ListenPidScan } from "../server/port-reclaim"; import { isLinkPort, JOIN_TUNNEL_PORT_MAX, JOIN_TUNNEL_PORT_MIN } from "../link/ports"; import { buildExecArgv, REMOTE_COMMAND_NOT_FOUND, remoteOcxArgv } from "../link/ssh-argv"; import { sshFailureHint, sshRunnerErrorHint, type SshRunner, type SshRunResult } from "../link/ssh-runner"; @@ -22,6 +23,7 @@ import type { OcxConnectedClientId } from "../types"; const JOIN_TUNNEL_READY_TIMEOUT_MS = 15_000; const JOIN_TUNNEL_POLL_MS = 100; +const JOIN_TUNNEL_SPAWN_GRACE_MS = 100; const JOIN_REVOKE_TIMEOUT_MS = 30_000; const JOIN_CONFIRM_TTL_MS = 5 * 60_000; const JOIN_PORT_ATTEMPTS = 32; @@ -74,6 +76,12 @@ export interface ClientLinkJoinDeps { hostname?: () => string; randomBytes?: (size: number) => Uint8Array; fetchImpl?: typeof fetch; + /** + * LISTEN-owner probe for the tunnel port; defaults to the netstat/lsof/ss scan. + * Receives the loopback address the tunnel binds so listeners on unrelated + * addresses do not confuse the readiness check. + */ + scanListenPids?: (port: number, address?: string) => ListenPidScan; spawnTunnel?: (spec: { linkId: string; alias: string; @@ -222,8 +230,10 @@ async function compensateStaleSidecar(deps: ClientLinkJoinDeps): Promise { } } +/** Wait for the live tunnel's authenticated readiness, retaining the deadline after failed ownership rechecks. */ async function waitForReady( deps: ClientLinkJoinDeps, + tunnel: ClientLinkTunnelHandle, port: number, key: string, ): Promise { @@ -231,13 +241,47 @@ async function waitForReady( const now = deps.now ?? Date.now; const sleep = deps.sleep ?? ((ms: number) => new Promise(resolve => setTimeout(resolve, ms))); const deadline = now() + JOIN_TUNNEL_READY_TIMEOUT_MS; + const tunnelExited = tunnel.exited.then(() => { throw new ClientLinkJoinError("join_tunnel_failed"); }); + await Promise.race([ + tunnelExited, + new Promise(resolve => setTimeout(resolve, JOIN_TUNNEL_SPAWN_GRACE_MS)), + ]); + const listenPids = deps.scanListenPids ?? scanListenPidsForAddress; + // The tunnel binds 127.0.0.1; a listener on a different loopback or interface address + // never receives our requests, so ownership is only judged among sockets that serve it. + const tunnelAddress = "127.0.0.1"; for (;;) { try { - const response = await fetchImpl(`http://127.0.0.1:${port}/readyz`, { - headers: { "x-opencodex-api-key": key }, - }); - if (response.status === 200) return; - if (response.status === 401) throw new ClientLinkJoinError("admission_failed"); + // A squatter answering the 401 challenge would otherwise collect the issued key: + // the only listener allowed a keyed request is the ssh process we spawned — it owns + // the port only after a successful bind, and ExitOnForwardFailure makes it exit when + // it cannot take the port. An unverifiable scan stays "not ready", never a pass. + const ownership = listenPids(port, tunnelAddress); + if (ownership.ok && ownership.pids.length === 1 && ownership.pids[0] === tunnel.pid) { + // Never follow redirects: a port occupant must not reroute the challenge, and a + // redirected keyed request would carry the issued key to an unrelated listener. + const probe = await Promise.race([ + tunnelExited, + fetchImpl(`http://127.0.0.1:${port}/readyz`, { redirect: "manual" }), + ]); + if (probe.status === 401) { + // Ownership can flip between the probe and the keyed request (a squatter + // takes the port after the tunnel dies). Re-scan in the same iteration and + // skip only the keyed request on failure, not the deadline check and sleep. + const recheck = listenPids(port, tunnelAddress); + if (recheck.ok && recheck.pids.length === 1 && recheck.pids[0] === tunnel.pid) { + const response = await Promise.race([ + tunnelExited, + fetchImpl(`http://127.0.0.1:${port}/readyz`, { + headers: { "x-opencodex-api-key": key }, + redirect: "manual", + }), + ]); + if (response.status === 200) return; + if (response.status === 401) throw new ClientLinkJoinError("admission_failed"); + } + } + } } catch (error) { if (error instanceof ClientLinkJoinError) throw error; } @@ -256,6 +300,7 @@ function requireConfirmedHost(deps: ClientLinkJoinDeps, alias: string): JoinConf return confirmed; } +/** Issue and enroll a confirmed Home link, compensating failures before committing the connection. */ export async function joinHome(deps: ClientLinkJoinDeps, input: { alias: string }): Promise<{ linkId: string; apiKeyId: string }> { const confirmed = requireConfirmedHost(deps, input.alias); await compensateStaleSidecar(deps); @@ -308,7 +353,7 @@ export async function joinHome(deps: ClientLinkJoinDeps, input: { alias: string configDir: deps.configDir, knownHostsFile: deps.knownHostsFile, }); - await waitForReady(deps, tunnelPort, issued.key); + await waitForReady(deps, tunnel, tunnelPort, issued.key); } catch (error) { const code = error instanceof ClientLinkJoinError ? error.code : "join_tunnel_failed"; await rollback(deps, issued.linkId, tunnel); @@ -316,22 +361,28 @@ export async function joinHome(deps: ClientLinkJoinDeps, input: { alias: string } try { + if (!tunnel) throw new ClientLinkJoinError("join_tunnel_failed"); const connect = deps.connect ?? connectClient; - await connect({ - serverUrl: `http://127.0.0.1:${tunnelPort}`, - managementUrl: `http://127.0.0.1:${tunnelPort}`, - credential: { kind: "link", apiKeyId: issued.apiKeyId, key: issued.key }, - transport: "link", - link: { tunnelPort, linkId: issued.linkId }, - selectedClients: deps.selectedClients ?? ["codex", "claude"], - managementTransport: "direct", - }, { - fetchImpl: deps.fetchImpl, - ...deps.connectDeps, - }); - } catch { + // Keep watching the tunnel until the connection commits: an exited tunnel + // must not let the issued key ride out to whatever next holds the port. + await Promise.race([ + tunnel.exited.then(() => { throw new ClientLinkJoinError("join_tunnel_failed"); }), + connect({ + serverUrl: `http://127.0.0.1:${tunnelPort}`, + managementUrl: `http://127.0.0.1:${tunnelPort}`, + credential: { kind: "link", apiKeyId: issued.apiKeyId, key: issued.key }, + transport: "link", + link: { tunnelPort, linkId: issued.linkId }, + selectedClients: deps.selectedClients ?? ["codex", "claude"], + managementTransport: "direct", + }, { + fetchImpl: deps.fetchImpl, + ...deps.connectDeps, + }), + ]); + } catch (error) { await rollback(deps, issued.linkId, tunnel); - throw new ClientLinkJoinError("join_connect_failed"); + throw new ClientLinkJoinError(error instanceof ClientLinkJoinError ? error.code : "join_connect_failed"); } await stopTunnel(tunnel); diff --git a/src/server/port-reclaim.ts b/src/server/port-reclaim.ts index 42fdc2305c2..97827cb60ae 100644 --- a/src/server/port-reclaim.ts +++ b/src/server/port-reclaim.ts @@ -18,6 +18,16 @@ export type ListenPidScan = | { ok: true; pids: number[] } | { ok: false; error?: string }; +/** One listening socket with its bound local address (host part only). */ +export interface ListenEntry { + pid: number; + address: string; +} + +export type ListenEntryScan = + | { ok: true; listeners: ListenEntry[] } + | { ok: false; error?: string }; + export type ReclaimListenPortOptions = WaitForPortOptions & { /** * When true AND `onlyKillPids` is a non-empty allowlist, those PIDs may be @@ -60,12 +70,49 @@ export type ReclaimListenPortOptions = WaitForPortOptions & { sleepMs?: (ms: number) => Promise; }; +/** Split `host:port`/`[v6]:port` on a numeric port boundary; returns the host part. */ +function listenHost(token: string): string { + const bracketed = /^(\[[0-9a-fA-F:.]+\]):/.exec(token); + if (bracketed) return bracketed[1].slice(1, -1).toLowerCase(); + // Only a trailing : is a port; a bare "::" or hostname wildcard has none. + const withPort = /^(.*):(\d+)$/.exec(token); + return (withPort ? withPort[1] : token).toLowerCase(); +} + +/** Normalize a listen-address host: strips brackets and the IPv4-mapped prefix. */ +export function normalizeListenAddress(token: string): string { + let host = listenHost(token); + if (host.startsWith("::ffff:")) host = host.slice(7); + return host; +} + +/** Normalize a bare bind address (no port): drops brackets, keeps bare IPv6 whole. */ +function bareListenAddress(address: string): string { + let host = address.replace(/^\[|\]$/g, "").toLowerCase(); + if (host.startsWith("::ffff:")) host = host.slice(7); + return host; +} + +const WILDCARD_LISTEN_HOSTS = new Set(["", "*", "0.0.0.0", "::"]); + /** - * Parse `netstat -ano` (Windows) / `netstat -anlp` listen lines for a port. - * Exported for unit tests. + * Whether a socket bound to `listenerAddress` also serves connections to `bound` — + * exact match, or a wildcard listener, or a wildcard `bound` (the caller listens on + * every address). IPv4-mapped IPv6 forms of the same address are equalized first. */ -export function parseListenPidsFromNetstat(output: string, port: number): number[] { - const pids = new Set(); +export function listenAddressServes(listenerAddress: string, bound: string): boolean { + const listener = normalizeListenAddress(listenerAddress); + const want = bareListenAddress(bound); + return WILDCARD_LISTEN_HOSTS.has(listener) || WILDCARD_LISTEN_HOSTS.has(want) + || listener === want; +} + +/** + * Parse `netstat -ano` (Windows) / `netstat -anlp` listen lines for a port, keeping + * each distinct PID/address pair. Exported for unit tests. + */ +export function parseListenEntriesFromNetstat(output: string, port: number): ListenEntry[] { + const entries = new Map(); const portSuffix = `:${port}`; for (const rawLine of output.split(/\r?\n/)) { const line = rawLine.trim(); @@ -88,9 +135,66 @@ export function parseListenPidsFromNetstat(output: string, port: number): number : unixPid ? Number(unixPid[1]) : NaN; - if (Number.isSafeInteger(pid) && pid > 0) pids.add(pid); + if (Number.isSafeInteger(pid) && pid > 0) { + const address = normalizeListenAddress(parts[localIdx]); + entries.set(`${pid}|${address}`, { pid, address }); + } } - return [...pids]; + return [...entries.values()]; +} + +/** Parse netstat LISTEN owners, deduplicating PIDs after preserving their addresses. */ +export function parseListenPidsFromNetstat(output: string, port: number): number[] { + return [...new Set(parseListenEntriesFromNetstat(output, port).map(entry => entry.pid))]; +} + +/** + * Parse `ss -Hltnp` rows for a port, keeping each distinct PID/address pair. A row + * without a `pid=` attribution (another user's socket) is dropped rather than + * reported unverifiable. Exported for unit tests. + */ +export function parseListenEntriesFromSs(output: string, port: number): ListenEntry[] { + const entries = new Map(); + const portSuffix = `:${port}`; + for (const rawLine of output.split(/\r?\n/)) { + const line = rawLine.trim(); + if (!/^LISTEN\b/i.test(line)) continue; + const parts = line.split(/\s+/); + // LISTEN users:(...) + const localIdx = parts.findIndex(part => part.endsWith(portSuffix) || part.endsWith(`]:${port}`)); + if (localIdx < 0) continue; + const pidMatch = /pid=(\d+)/.exec(line); + const pid = pidMatch ? Number(pidMatch[1]) : NaN; + if (Number.isSafeInteger(pid) && pid > 0) { + const address = normalizeListenAddress(parts[localIdx]); + entries.set(`${pid}|${address}`, { pid, address }); + } + } + return [...entries.values()]; +} + +/** + * Parse `lsof -nP -iTCP: -sTCP:LISTEN` output (without -t), keeping each + * distinct PID/address pair. The NAME column is the last address token, optionally + * followed by `(LISTEN)`; skip the header and nonnumeric PIDs. Exported for tests. + */ +export function parseListenEntriesFromLsof(output: string, port: number): ListenEntry[] { + const entries = new Map(); + const portSuffix = `:${port}`; + for (const rawLine of output.split(/\r?\n/)) { + const line = rawLine.trim(); + if (!line || /^COMMAND\b/.test(line)) continue; + const parts = line.split(/\s+/); + const pid = /^\d+$/.test(parts[1] ?? "") ? Number(parts[1]) : NaN; + if (!Number.isSafeInteger(pid) || pid <= 0) continue; + let addressIdx = parts.length - 1; + if (/^\(.*\)$/.test(parts[addressIdx] ?? "")) addressIdx -= 1; + const address = parts[addressIdx] ?? ""; + if (!address.endsWith(portSuffix) && !address.endsWith(`]:${port}`)) continue; + const normalized = normalizeListenAddress(address); + entries.set(`${pid}|${normalized}`, { pid, address: normalized }); + } + return [...entries.values()]; } function normalizeListenPidScan(result: ListenPidScan | number[]): ListenPidScan { @@ -121,50 +225,84 @@ function readWindowsNetstatAno(): string { } /** - * Scan for PIDs currently LISTENing on `port`. - * Distinguishes probe failure (`ok: false`) from a successful empty result. + * Scan for the sockets currently LISTENing on `port`, with each listener's bound + * local address. Distinguishes probe failure (`ok: false`) from a successful empty + * result. POSIX backends are tried in order — `lsof`, `ss` (iproute2, the only + * scanner on minimal Linux installs), then `netstat` — and a missing scanner falls + * through to the next instead of failing the scan. */ -export function scanListenPids(port: number): ListenPidScan { +export function scanListenEntries(port: number): ListenEntryScan { if (!Number.isFinite(port) || port <= 0 || port > 65535) { return { ok: false, error: "invalid port" }; } + const scanned = Math.trunc(port); try { if (process.platform === "win32") { - return { ok: true, pids: parseListenPidsFromNetstat(readWindowsNetstatAno(), port) }; + return { ok: true, listeners: parseListenEntriesFromNetstat(readWindowsNetstatAno(), scanned) }; } + const errors: string[] = []; try { - const output = execFileSync("lsof", ["-nP", `-iTCP:${port}`, "-sTCP:LISTEN", "-t"], { + const output = execFileSync("lsof", ["-nP", `-iTCP:${scanned}`, "-sTCP:LISTEN"], { encoding: "utf-8", stdio: ["ignore", "pipe", "ignore"], timeout: 3000, }); - return { - ok: true, - pids: output - .split(/\r?\n/) - .map(line => Number(line.trim())) - .filter(pid => Number.isSafeInteger(pid) && pid > 0), - }; - } catch (lsofErr) { - try { - const output = execFileSync("netstat", ["-anlp"], { - encoding: "utf-8", - stdio: ["ignore", "pipe", "ignore"], - timeout: 3000, - }); - return { ok: true, pids: parseListenPidsFromNetstat(output, Math.trunc(port)) }; - } catch (netstatErr) { - return { - ok: false, - error: `lsof/netstat unavailable: ${String(lsofErr)} / ${String(netstatErr)}`, - }; - } + return { ok: true, listeners: parseListenEntriesFromLsof(output, scanned) }; + } catch (error) { + errors.push(`lsof: ${String(error)}`); + } + try { + const output = execFileSync("ss", ["-Hltnp"], { + encoding: "utf-8", + stdio: ["ignore", "pipe", "ignore"], + timeout: 3000, + }); + return { ok: true, listeners: parseListenEntriesFromSs(output, scanned) }; + } catch (error) { + errors.push(`ss: ${String(error)}`); + } + try { + const output = execFileSync("netstat", ["-anlp"], { + encoding: "utf-8", + stdio: ["ignore", "pipe", "ignore"], + timeout: 3000, + }); + return { ok: true, listeners: parseListenEntriesFromNetstat(output, scanned) }; + } catch (error) { + errors.push(`netstat: ${String(error)}`); } + return { ok: false, error: `no listener scanner available (${errors.join(" / ")})` }; } catch (error) { return { ok: false, error: String(error) }; } } +/** + * Scan for PIDs currently LISTENing on `port`. + * Distinguishes probe failure (`ok: false`) from a successful empty result. + */ +export function scanListenPids(port: number): ListenPidScan { + const scan = scanListenEntries(port); + if (!scan.ok) return { ok: false, error: scan.error }; + return { ok: true, pids: [...new Set(scan.listeners.map(entry => entry.pid))] }; +} + +/** + * PIDs LISTENing on `port` that actually serve `address`: listeners bound to that + * exact address plus wildcards (0.0.0.0/::). A listener on a different loopback or + * interface address (e.g. 127.0.0.2 while the tunnel binds 127.0.0.1) never receives + * the connection and must not block or qualify a readiness check. + */ +export function scanListenPidsForAddress(port: number, address = "127.0.0.1"): ListenPidScan { + const scan = scanListenEntries(port); + if (!scan.ok) return { ok: false, error: scan.error }; + const pids = new Set(); + for (const entry of scan.listeners) { + if (listenAddressServes(entry.address, address)) pids.add(entry.pid); + } + return { ok: true, pids: [...pids] }; +} + /** Best-effort PIDs currently LISTENing on `port`. Empty on probe failure. */ export function listListenPids(port: number): number[] { const scan = scanListenPids(port); diff --git a/structure/remote-link.md b/structure/remote-link.md index e4edc2419e0..cfad32f1cad 100644 --- a/structure/remote-link.md +++ b/structure/remote-link.md @@ -18,6 +18,10 @@ Every remote `ocx` call goes through `remoteOcxArgv`, which runs `sh -c` with a `src/server/management/link-routes.ts` accepts `POST /api/link/join` with exactly `{ "alias": string }`. The route admits the same dashboard sessions as the Home-side routes (see [Dashboard admission](#dashboard-admission)), so a standalone computer turns itself into a Child from its own dashboard. A Tailscale identity session receives `403 tailscale_session_refused`, any other caller `403 forbidden`, a runtime that is not standalone `409 standalone_required`, and a standalone whose live listener port (`resolveListenPort` in `src/server/management/system-restart.ts`) is not its configured `port`, or cannot be determined, `409 join_port_mismatch`, because the client runtime it restarts into binds exactly the configured port. These gates run before link state is read and before any SSH. The alias must have a confirmed, unexpired host entry in the same route state. Before choosing a port or issuing a new link, a valid stale client sidecar is compensated over SSH unless the machine is already connected to that link; a successful revoke clears the sidecar, while a failed revoke preserves it and returns `join_rollback_failed` with the link id. A corrupt sidecar is left for the next successful write. A successful join issues the Home link through SSH, records the client sidecar, starts the client tunnel and connects the client, then returns `202 { "linkId": string, "alias": string, "restarting": true }`. +During enrollment, `src/client/link-join.ts` watches the spawned SSH tunnel through its 100 ms spawn grace, readiness checks and the connection attempt. Before an unauthenticated `/readyz` probe and again before the keyed request, the only LISTEN owner serving `127.0.0.1:` must be that tunnel PID. Both requests use `redirect: "manual"`; only a 401 challenge permits the keyed request. A failed, empty, foreign or ambiguous ownership recheck withholds the key and reaches the same 15-second deadline check and up-to-100 ms polling delay as any other not-ready iteration. Repeated recheck failures therefore reach rollback instead of bypassing it. The deadline is checked between operations, not an independent per-fetch cancellation timer. An observed tunnel exit winning the readiness or connection race fails the join and runs compensation. + +The enrollment scanner in `src/server/port-reclaim.ts` retains each distinct normalized `(PID, bound address)` pair from Windows `netstat` or the POSIX `lsof`, `ss`, then `netstat` fallback chain. It filters entries for the requested loopback address before deduplicating PIDs, so another socket owned by the same process cannot overwrite the relevant listener. Duplicate rows and IPv4-mapped aliases of the same address still collapse, and the PID-only API continues to return unique PIDs. This enrollment scan does not replace the runtime supervisor's asynchronous ownership check described under [Client link transport](#client-link-transport); the check-to-connect race described there remains. + `src/client/link-state.ts` stores `/link/client-link.json` with mode 0600. The sidecar contains exactly `alias`, `hubHostKeyFingerprint`, `tunnelPort`, `peerListenerPort` and `linkId`; it contains no key. The client tunnel port uses `MIN_LINK_PORT = 1024` through `MAX_LINK_PORT = 65535` and `isLinkPort`; the Home listener port keeps its existing 1–65535 contract. A dashboard join picks a free port at random from `JOIN_TUNNEL_PORT_MIN = 20000` through `JOIN_TUNNEL_PORT_MAX = 29999` (`chooseJoinTunnelPort` in `src/client/link-join.ts`), below the macOS, Windows and Linux ephemeral ranges, so an outgoing connection rarely holds the port when the tunnel comes back after a reboot. The persisted port of an existing link is never rewritten. `src/client/link-tunnel.ts` owns the client `ssh -L 127.0.0.1::127.0.0.1:` process. A client runtime starts that supervisor when link transport and a matching sidecar are present, and it drives the tunnel with `CLIENT_TUNNEL_RETRY_POLICY`, so the Child reconnects by itself after sleep, an outage or a crash. A spawned or respawned tunnel, whether connecting, reconnecting or retrying from failed, is promoted to connected only when a keyed `GET http://127.0.0.1:/readyz` (the link key in `x-opencodex-api-key`, `cache: "no-store"`) proves the link: a 200, or a 503 whose body (read up to 4 KiB) carries `service: "opencodex"`. The Home's link listener answers 401 before it reaches `/readyz`, so that 503 only means the Home's own start-up readiness is pending or failed, which relayed requests do not depend on. Until then the probe backs off from one to five seconds, or up to 30 seconds while the link reads failed; while a request is held it runs on every one-second check instead. One probe runs at a time, detached from the check, and `stop()` or the end of the tunnel it probes aborts it, so a slow Home never delays noticing a disconnect or stopping. While connected the same probe runs every 30 seconds and is display-only: a 401 or 403 reports `probe: "unauthorized"`, the readiness 503 `probe: "home_not_ready"`, and any other failure `probe: "home_unreachable"` in the supervisor status, which the Child's `GET /api/link/status` shows as the child `reason`; it never cuts the tunnel. The key comes from the runtime's cached key source, so no probe reads the token file. The supervisor's one-second check is an unref'd interval that stats the sidecar and `config.json` and parses one again only after it changed. It stops the tunnel and schedules the existing standalone recycle when the connection is no longer connected with link transport, the link id no longer matches, or the sidecar disappears; an unreadable sidecar or connection state acts only after three consecutive checks, so one read during a write never ends a healthy link. Normal shutdown, including recycle, stops the client supervisor before the client listener; it sends TERM, waits at most five seconds, then sends KILL. @@ -58,4 +62,4 @@ Codex keeps the standalone loopback routing: `routingTarget` in `src/client/conn > Decision record: [ADR-6032](decisions/ADR-6032-link-relay-credential-boundary.md) -Regression coverage lives in `tests/clients/link-ssh-argv.test.ts`, `tests/clients/link-ssh-config.test.ts`, `tests/clients/link-tunnel-state.test.ts`, `tests/clients/link-store.test.ts`, `tests/clients/link-boundary.test.ts`, `tests/clients/link-routes.test.ts`, `tests/clients/client-link-connect.test.ts`, `tests/clients/client-link-relay.test.ts`, `tests/clients/client-machine-listener.test.ts`, `tests/clients/client-link-status.test.ts`, `tests/clients/client-link-runtime.test.ts`, `tests/codex-integration/injection-link-websocket.test.ts`, `tests/clients/link-supervisor.test.ts`, `tests/clients/link-status-projection.test.ts`, `tests/clients/link-admission-wait.test.ts`, `tests/clients/link-fingerprint.test.ts`, `tests/cli/cli-link.test.ts`, `tests/server/link-management-routes.test.ts`, `tests/server/link-join-route.test.ts`, `tests/server/link-listener-lifecycle.test.ts`, `tests/clients/client-link-teardown.test.ts` and `gui/tests/remote-link.test.tsx`. +Regression coverage lives in `tests/clients/link-ssh-argv.test.ts`, `tests/clients/link-ssh-config.test.ts`, `tests/clients/link-tunnel-state.test.ts`, `tests/clients/link-store.test.ts`, `tests/clients/link-boundary.test.ts`, `tests/clients/link-routes.test.ts`, `tests/clients/client-link-connect.test.ts`, `tests/clients/client-link-relay.test.ts`, `tests/clients/client-machine-listener.test.ts`, `tests/clients/client-link-status.test.ts`, `tests/clients/client-link-runtime.test.ts`, `tests/codex-integration/injection-link-websocket.test.ts`, `tests/clients/link-supervisor.test.ts`, `tests/clients/link-status-projection.test.ts`, `tests/clients/link-admission-wait.test.ts`, `tests/clients/link-fingerprint.test.ts`, `tests/cli/cli-link.test.ts`, `tests/server/link-management-routes.test.ts`, `tests/server/link-join-route.test.ts`, `tests/server/port-reclaim.test.ts`, `tests/server/link-listener-lifecycle.test.ts`, `tests/clients/client-link-teardown.test.ts` and `gui/tests/remote-link.test.tsx`. diff --git a/tests/server/link-join-route.test.ts b/tests/server/link-join-route.test.ts index 653e749bf37..9ea2b8c26c1 100644 --- a/tests/server/link-join-route.test.ts +++ b/tests/server/link-join-route.test.ts @@ -1,5 +1,7 @@ import { describe, expect, test, spyOn } from "bun:test"; import { chooseJoinTunnelPort, ClientLinkJoinError, joinHome, type ClientLinkJoinDeps } from "../../src/client/link-join"; +import { spawnClientLinkTunnel } from "../../src/client/link-tunnel"; +import type { ListenPidScan } from "../../src/server/port-reclaim"; import { isLinkPort, JOIN_TUNNEL_PORT_MAX, JOIN_TUNNEL_PORT_MIN } from "../../src/link/ports"; import { handleLinkRoutes, type LinkRouteState } from "../../src/server/management/link-routes"; import type { ManagementContext } from "../../src/server/management/context"; @@ -14,6 +16,7 @@ function isWrappedIssue(argv: readonly string[]): boolean { return (argv.at(-1) ?? "").startsWith(`${quoteRemote(remoteOcxArgv(["link", "issue", "--alias"]))} `); } +/** Select only the wrapped revoke command, not other SSH traffic. */ function revokeCalls(calls: readonly string[][]): string[][] { return calls.filter(argv => argv.at(-1) === REVOKE_COMMAND); } @@ -21,6 +24,7 @@ const API_KEY_ID = "link-key-1"; const KEY = `ocx_data_${"a".repeat(40)}`; const FINGERPRINT = `SHA256:${"a".repeat(32)}`; +/** Record issue and compensation calls without starting an SSH process. */ function runnerFor(calls: string[][], issueResult = true): SshRunner { return { run: async argv => { @@ -41,16 +45,29 @@ function runnerFor(calls: string[][], issueResult = true): SshRunner { }; } +/** Keep the fake tunnel alive until its owner stops it. */ function tunnelFor(order: string[]) { return { pid: 123, - exited: Promise.resolve(0), + exited: new Promise(() => {}), stop: async () => { order.push("stop-tunnel"); }, }; } +/** Return the link-auth challenge before accepting the issued key. */ +function challengedFetch(order?: string[]) { + return async (_input: RequestInfo | URL, init?: RequestInit) => { + const authed = new Headers(init?.headers).get("x-opencodex-api-key") === KEY; + order?.push(authed ? "readyz:key" : "readyz:probe"); + return new Response(null, { status: authed ? 200 : 401 }); + }; +} + +/** Supply isolated join dependencies and attribute the listener to the fake tunnel. */ function joinDeps(overrides: Partial = {}): ClientLinkJoinDeps { - const calls = overrides.runner ? [] : []; + const calls: string[][] = []; + let tunnelPid = 0; + const spawn = overrides.spawnTunnel ?? spawnClientLinkTunnel; return { runner: overrides.runner ?? runnerFor(calls), knownHostsFile: "/tmp/ocx-known-hosts", @@ -61,7 +78,14 @@ function joinDeps(overrides: Partial = {}): ClientLinkJoinDe hostname: () => "client-host", readSidecar: () => null, readConnectionState: () => ({ kind: "disconnected" }), + scheduleRestart: () => {}, ...overrides, + spawnTunnel: (spec, spawnDeps) => { + const handle = spawn(spec, spawnDeps); + tunnelPid = handle.pid; + return handle; + }, + scanListenPids: overrides.scanListenPids ?? (() => ({ ok: true, pids: [tunnelPid] })), }; } @@ -258,10 +282,8 @@ describe("client initiated link join", () => { hostname: () => "client-host", writeState: state => { order.push("write-state"); Object.assign(sidecar, state); }, spawnTunnel: () => { order.push("spawn-tunnel"); return tunnelFor(order); }, - fetchImpl: async (_input, init) => { - order.push(`readyz:${new Headers(init?.headers).get("x-opencodex-api-key") === KEY ? "key" : "missing"}`); - return new Response(null, { status: 200 }); - }, + scanListenPids: () => ({ ok: true, pids: [123] }), + fetchImpl: challengedFetch(order), connect: (async () => { order.push("connect"); }) as typeof import("../../src/client/connect").connectClient, scheduleRestart: () => { order.push("restart"); }, }, { alias: "home" })), @@ -271,7 +293,7 @@ describe("client initiated link join", () => { expect(response?.status).toBe(202); expect(responseBody).toEqual({ linkId: LINK_ID, alias: "home", restarting: true }); expect(sidecar).toMatchObject({ linkId: LINK_ID, tunnelPort: 23456, peerListenerPort: 45678 }); - expect(order).toEqual(["write-state", "spawn-tunnel", "readyz:key", "connect", "stop-tunnel", "restart"]); + expect(order).toEqual(["write-state", "spawn-tunnel", "readyz:probe", "readyz:key", "connect", "stop-tunnel", "restart"]); expect(isWrappedIssue(calls[0] ?? [])).toBe(true); expect(calls[0]?.at(-1)?.endsWith(" '--json'")).toBe(true); }); @@ -297,7 +319,7 @@ describe("client initiated link join", () => { choosePort: undefined, writeState: state => { sidecarPort = state.tunnelPort; }, spawnTunnel: () => tunnelFor([]), - fetchImpl: async () => new Response(null, { status: 200 }), + fetchImpl: challengedFetch(), connect: (async () => {}) as typeof import("../../src/client/connect").connectClient, scheduleRestart: () => {}, }), { alias: "home" }); @@ -330,7 +352,7 @@ describe("client initiated link join", () => { sleep: async () => {}, writeState: () => {}, clearState: () => { cleared += 1; }, - spawnTunnel: () => ({ pid: 1, exited: Promise.resolve(0), stop: async () => { stopped += 1; } }), + spawnTunnel: () => ({ pid: 1, exited: new Promise(() => {}), stop: async () => { stopped += 1; } }), fetchImpl: async () => readiness === "unauthorized" ? new Response(null, { status: 401 }) : new Response(null, { status: 503 }), }); await expect(joinHome(deps, { alias: "home" })).rejects.toMatchObject({ @@ -342,6 +364,221 @@ describe("client initiated link join", () => { } }); + test("never sends the issued key to a listener that skips the link-auth challenge", async () => { + const calls: string[][] = []; + let keyedFetches = 0; + let ticks = 0; + let stopped = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + now: () => (ticks++ === 0 ? 0 : 15_002 * ticks), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => ({ pid: 1, exited: new Promise(() => {}), stop: async () => { stopped += 1; } }), + fetchImpl: async (_input, init) => { + if (new Headers(init?.headers).has("x-opencodex-api-key")) keyedFetches += 1; + return new Response(null, { status: 200 }); + }, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(keyedFetches).toBe(0); + expect(stopped).toBe(1); + expect(revokeCalls(calls)).toHaveLength(1); + }); + + test("a squatter answering the 401 challenge never receives the issued key", async () => { + const calls: string[][] = []; + let keyedFetches = 0; + let ticks = 0; + let stopped = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + now: () => (ticks++ === 0 ? 0 : 15_002 * ticks), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => ({ pid: 123, exited: new Promise(() => {}), stop: async () => { stopped += 1; } }), + // A live SSH process alone does not prove ownership of the listener. + scanListenPids: () => ({ ok: true, pids: [999] }), + fetchImpl: async (_input, init) => { + if (new Headers(init?.headers).has("x-opencodex-api-key")) keyedFetches += 1; + return new Response(null, { status: 401 }); + }, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(keyedFetches).toBe(0); + expect(stopped).toBe(1); + expect(revokeCalls(calls)).toHaveLength(1); + }); + + test("the readiness scan is scoped to the tunnel's loopback address", async () => { + const seenAddresses: Array = []; + const order: string[] = []; + await joinHome(joinDeps({ + runner: runnerFor([]), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => tunnelFor(order), + scanListenPids: (_port, address) => { + seenAddresses.push(address); + return { ok: true, pids: [123] }; + }, + fetchImpl: challengedFetch(order), + connect: (async () => {}) as never, + scheduleRestart: () => {}, + }), { alias: "home" }); + expect(seenAddresses.length).toBeGreaterThan(0); + for (const address of seenAddresses) expect(address).toBe("127.0.0.1"); + }); + + test("a port flip between the probe and the keyed request never receives the key", async () => { + const calls: string[][] = []; + let keyedFetches = 0; + let ticks = 0; + let scans = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + now: () => (ticks++ === 0 ? 0 : 15_002 * ticks), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => ({ pid: 123, exited: new Promise(() => {}), stop: async () => {} }), + scanListenPids: () => ({ ok: true, pids: scans++ === 0 ? [123] : [999] }), + fetchImpl: async (_input, init) => { + if (new Headers(init?.headers).has("x-opencodex-api-key")) keyedFetches += 1; + return new Response(null, { status: 401 }); + }, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(keyedFetches).toBe(0); + expect(revokeCalls(calls)).toHaveLength(1); + }); + + test("a redirect on the readiness probe is never followed with the issued key", async () => { + const calls: string[][] = []; + const redirects: Array = []; + let keyedFetches = 0; + let ticks = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + now: () => (ticks++ === 0 ? 0 : 15_002 * ticks), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => ({ pid: 123, exited: new Promise(() => {}), stop: async () => {} }), + fetchImpl: async (_input, init) => { + redirects.push(init?.redirect); + if (new Headers(init?.headers).has("x-opencodex-api-key")) keyedFetches += 1; + return new Response(null, { status: 302, headers: { location: "http://169.254.1.1/fake-readyz" } }); + }, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(redirects).toEqual(["manual"]); + expect(keyedFetches).toBe(0); + expect(revokeCalls(calls)).toHaveLength(1); + }); + + test("a tunnel that exits during connect cannot commit the connection", async () => { + const calls: string[][] = []; + let releaseExit!: (code: number) => void; + const exited = new Promise(resolve => { releaseExit = resolve; }); + let stopped = 0; + let connectCommitted = false; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => ({ pid: 1, exited, stop: async () => { stopped += 1; } }), + fetchImpl: challengedFetch(), + connect: (async () => { releaseExit(255); await new Promise(() => {}); connectCommitted = true; }) as typeof import("../../src/client/connect").connectClient, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(connectCommitted).toBe(false); + expect(stopped).toBe(1); + expect(revokeCalls(calls)).toHaveLength(1); + }); + + test("does not disclose the issued key when the tunnel exits during its spawn grace", async () => { + const calls: string[][] = []; + let fetches = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + writeState: () => {}, + clearState: () => {}, + spawnTunnel: () => ({ pid: 1, exited: Promise.resolve(255), stop: async () => {} }), + fetchImpl: async () => { + fetches += 1; + return new Response(null, { status: 200 }); + }, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(fetches).toBe(0); + expect(revokeCalls(calls)).toHaveLength(1); + }); + + for (const { name, recheck } of [ + { name: "unavailable", recheck: { ok: false, error: "scanner unavailable" } }, + { name: "empty", recheck: { ok: true, pids: [] } }, + { name: "foreign", recheck: { ok: true, pids: [999] } }, + { name: "ambiguous", recheck: { ok: true, pids: [123, 999] } }, + ] satisfies Array<{ name: string; recheck: ListenPidScan }>) { + test(`repeated ${name} rechecks reach the readiness deadline and revoke the key`, async () => { + const calls: string[][] = []; + const sleeps: number[] = []; + let clock = 1; + let scans = 0; + let keyedFetches = 0; + let stopped = 0; + let cleared = 0; + let connected = 0; + let restarted = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + now: () => clock, + sleep: async ms => { sleeps.push(ms); clock += ms; }, + writeState: () => {}, + clearState: () => { cleared += 1; }, + spawnTunnel: () => ({ pid: 123, exited: new Promise(() => {}), stop: async () => { stopped += 1; } }), + scanListenPids: () => { + // Terminate the broken implementation without hanging the test runner. + // Its bypassed deadline produces the wrong error, so this is not a pass. + if (++scans > 400) throw new ClientLinkJoinError("admission_failed"); + return scans % 2 === 1 ? { ok: true, pids: [123] } : recheck; + }, + fetchImpl: async (_input, init) => { + if (new Headers(init?.headers).has("x-opencodex-api-key")) keyedFetches += 1; + return new Response(null, { status: 401 }); + }, + connect: (async () => { connected += 1; }) as typeof import("../../src/client/connect").connectClient, + scheduleRestart: () => { restarted += 1; }, + }), { alias: "home" })).rejects.toMatchObject({ code: "join_tunnel_failed" }); + expect(clock).toBe(15_001); + expect(sleeps).toHaveLength(150); + expect(sleeps.every(ms => ms === 100)).toBe(true); + expect(scans).toBe(302); + expect(keyedFetches).toBe(0); + expect(connected).toBe(0); + expect(restarted).toBe(0); + expect(stopped).toBe(1); + expect(cleared).toBe(1); + expect(revokeCalls(calls)).toHaveLength(1); + }); + } + + test("a transient failed recheck polls before retrying and can still join", async () => { + const calls: string[][] = []; + const order: string[] = []; + let clock = 1; + let scans = 0; + await expect(joinHome(joinDeps({ + runner: runnerFor(calls), + now: () => clock, + sleep: async ms => { clock += ms; order.push(`sleep:${ms}`); }, + writeState: () => {}, + spawnTunnel: () => tunnelFor(order), + scanListenPids: () => ++scans === 2 + ? { ok: false, error: "transient" } + : { ok: true, pids: [123] }, + fetchImpl: challengedFetch(order), + connect: (async () => { order.push("connect"); }) as typeof import("../../src/client/connect").connectClient, + scheduleRestart: () => { order.push("restart"); }, + }), { alias: "home" })).resolves.toEqual({ linkId: LINK_ID, apiKeyId: API_KEY_ID }); + expect(scans).toBe(4); + expect(order).toEqual(["readyz:probe", "sleep:100", "readyz:probe", "readyz:key", "connect", "stop-tunnel", "restart"]); + expect(revokeCalls(calls)).toHaveLength(0); + }); + test("rolls back on connect failure and never exposes the issued key", async () => { const calls: string[][] = []; const logs = spyOn(console, "log").mockImplementation(() => {}); @@ -351,7 +588,7 @@ describe("client initiated link join", () => { writeState: () => {}, clearState: () => {}, spawnTunnel: () => tunnelFor([]), - fetchImpl: async () => new Response(null, { status: 200 }), + fetchImpl: challengedFetch(), connect: (async () => { throw new Error(`connect failed ${KEY}`); }) as typeof import("../../src/client/connect").connectClient, }), { alias: "home" })).rejects.toMatchObject({ code: "join_connect_failed" }); } finally { @@ -392,8 +629,8 @@ describe("client initiated link join", () => { readSidecar: () => sidecarPresent ? sidecar : null, writeState: value => { sidecarPresent = true; Object.assign(sidecar, value); }, clearState: () => { sidecarPresent = false; }, - spawnTunnel: () => ({ pid: 1, exited: Promise.resolve(0), stop: async () => {} }), - fetchImpl: async () => new Response(null, { status: 200 }), + spawnTunnel: () => ({ pid: 1, exited: new Promise(() => {}), stop: async () => {} }), + fetchImpl: challengedFetch(), connect: (async () => { throw new Error("connect failed"); }) as typeof import("../../src/client/connect").connectClient, }); await expect(joinHome(base, { alias: "home" })).rejects.toMatchObject({ code: "join_rollback_failed", linkId: LINK_ID }); @@ -426,7 +663,8 @@ describe("client initiated link join", () => { writeState: state => { sidecar = { ...state }; }, clearState: () => { cleared = true; }, spawnTunnel: () => tunnelFor([]), - fetchImpl: async () => new Response(null, { status: 200 }), + scanListenPids: () => ({ ok: true, pids: [123] }), + fetchImpl: challengedFetch(), connect: (async () => { connected = true; }) as typeof import("../../src/client/connect").connectClient, scheduleRestart: () => { throw new Error("restart unavailable"); }, }, input)) as typeof import("../../src/client/link-join").joinHome, diff --git a/tests/server/port-reclaim.test.ts b/tests/server/port-reclaim.test.ts index b11f45e5af4..be63b0c449a 100644 --- a/tests/server/port-reclaim.test.ts +++ b/tests/server/port-reclaim.test.ts @@ -1,5 +1,13 @@ import { describe, expect, spyOn, test } from "bun:test"; +import { createServer } from "node:net"; +import * as childProcess from "node:child_process"; import { + listenAddressServes, + normalizeListenAddress, + parseListenEntriesFromLsof, + parseListenEntriesFromNetstat, + parseListenEntriesFromSs, + scanListenPidsForAddress, ownsIpv4LoopbackListener, parseIpv4LoopbackListenPidsFromNetstat, parseProcLoopbackListenInodes, @@ -126,6 +134,167 @@ describe("exact IPv4 loopback listener ownership", () => { }); }); +describe("listen-entry parsers keep the bound address", () => { + test("netstat entries report each listener's local address", () => { + const output = [ + "tcp 0 0 127.0.0.1:10100 0.0.0.0:* LISTEN 4242/bun", + "tcp 0 0 127.0.0.2:10100 0.0.0.0:* LISTEN 7777/foreign", + "tcp 0 0 127.0.0.1:22 0.0.0.0:* LISTEN 1/sshd", + ].join("\n"); + expect(parseListenEntriesFromNetstat(output, 10100)).toEqual([ + { pid: 4242, address: "127.0.0.1" }, + { pid: 7777, address: "127.0.0.2" }, + ]); + }); + + test("ss -Hltnp rows report address and pid; unattributed rows are dropped", () => { + const output = [ + "LISTEN 0 128 127.0.0.1:10100 0.0.0.0:* users:((\"bun\",pid=4242,fd=20))", + "LISTEN 0 128 127.0.0.2:10100 0.0.0.0:* users:((\"foreign\",pid=7777,fd=6))", + "LISTEN 0 128 127.0.0.1:10100 0.0.0.0:*", + "LISTEN 0 511 *:22 *:* users:((\"sshd\",pid=1,fd=3))", + ].join("\n"); + expect(parseListenEntriesFromSs(output, 10100)).toEqual([ + { pid: 4242, address: "127.0.0.1" }, + { pid: 7777, address: "127.0.0.2" }, + ]); + }); + + test("lsof NAME column supplies the bound address", () => { + const output = [ + "COMMAND PID USER FD TYPE DEVICE SIZE/OFF NODE NAME", + "bun 4242 devin 20u IPv4 0xdeadbeef 0t0 TCP 127.0.0.1:10100 (LISTEN)", + "other 7777 devin 21u IPv4 0xdeadbeef 0t0 TCP 127.0.0.2:10100 (LISTEN)", + ].join("\n"); + expect(parseListenEntriesFromLsof(output, 10100)).toEqual([ + { pid: 4242, address: "127.0.0.1" }, + { pid: 7777, address: "127.0.0.2" }, + ]); + }); + + test("address matching treats wildcards as serving any bound address", () => { + expect(listenAddressServes("127.0.0.1", "127.0.0.1")).toBe(true); + expect(listenAddressServes("127.0.0.2", "127.0.0.1")).toBe(false); + expect(listenAddressServes("0.0.0.0", "127.0.0.1")).toBe(true); + expect(listenAddressServes("*", "127.0.0.1")).toBe(true); + expect(listenAddressServes("::", "127.0.0.1")).toBe(true); + expect(listenAddressServes("[::1]:443", "::1")).toBe(true); + expect(normalizeListenAddress("::ffff:127.0.0.1")).toBe("127.0.0.1"); + expect(listenAddressServes("::ffff:127.0.0.1", "127.0.0.1")).toBe(true); + }); + + const multiAddressCases = [ + { + name: "Windows netstat", parse: parseListenEntriesFromNetstat, + rows: [ + "TCP 127.0.0.1:10100 0.0.0.0:0 LISTENING 4242", + "TCP 127.0.0.2:10100 0.0.0.0:0 LISTENING 4242", + "TCP [::ffff:127.0.0.1]:10100 [::]:0 LISTENING 4242", + ], + }, + { + name: "POSIX netstat", parse: parseListenEntriesFromNetstat, + rows: [ + "tcp 0 0 127.0.0.1:10100 0.0.0.0:* LISTEN 4242/ssh", + "tcp 0 0 127.0.0.2:10100 0.0.0.0:* LISTEN 4242/ssh", + "tcp 0 0 127.0.0.1:10100 0.0.0.0:* LISTEN 4242/ssh", + ], + }, + { + name: "ss", parse: parseListenEntriesFromSs, + rows: [ + 'LISTEN 0 128 127.0.0.1:10100 0.0.0.0:* users:(("ssh",pid=4242,fd=3))', + 'LISTEN 0 128 127.0.0.2:10100 0.0.0.0:* users:(("ssh",pid=4242,fd=4))', + 'LISTEN 0 128 [::ffff:127.0.0.1]:10100 [::]:* users:(("ssh",pid=4242,fd=5))', + ], + }, + { + name: "lsof", parse: parseListenEntriesFromLsof, + rows: [ + "ssh 4242 user 3u IPv4 0x1 0t0 TCP 127.0.0.1:10100 (LISTEN)", + "ssh 4242 user 4u IPv4 0x2 0t0 TCP 127.0.0.2:10100 (LISTEN)", + "ssh 4242 user 5u IPv6 0x3 0t0 TCP [::ffff:127.0.0.1]:10100 (LISTEN)", + ], + }, + ]; + for (const { name, parse, rows } of multiAddressCases) { + for (const reverse of [false, true]) { + test(`${name} retains all same-PID addresses with reverse=${reverse}`, () => { + const ordered = reverse ? [...rows].reverse() : rows; + const entries = parse([...ordered, ordered[0]].join("\n"), 10100); + expect([...entries].sort((a, b) => a.address.localeCompare(b.address))).toEqual([ + { pid: 4242, address: "127.0.0.1" }, + { pid: 4242, address: "127.0.0.2" }, + ]); + for (const address of ["127.0.0.1", "127.0.0.2"]) { + expect(entries.filter(entry => listenAddressServes(entry.address, address)).map(entry => entry.pid)).toEqual([4242]); + } + }); + } + } + + test("the PID-only netstat API still deduplicates multiple addresses", () => { + expect(parseListenPidsFromNetstat(multiAddressCases[0]!.rows.join("\n"), 10100)).toEqual([4242]); + }); + + test("the address-scoped scanner filters before deduplicating same-PID listeners", () => { + const fixture = process.platform === "win32" ? multiAddressCases[0]! : multiAddressCases[3]!; + for (const reverse of [false, true]) { + const rows = reverse ? [...fixture.rows].reverse() : fixture.rows; + const scan = spyOn(childProcess, "execFileSync").mockImplementation(() => rows.join("\n")); + try { + expect(scanListenPidsForAddress(10100)).toEqual({ ok: true, pids: [4242] }); + expect(scanListenPidsForAddress(10100, "127.0.0.1")).toEqual({ ok: true, pids: [4242] }); + expect(scanListenPidsForAddress(10100, "127.0.0.2")).toEqual({ ok: true, pids: [4242] }); + expect(scanListenPidsForAddress(10100, "127.0.0.3")).toEqual({ ok: true, pids: [] }); + expect(scanListenPidsForAddress(10100, "0.0.0.0")).toEqual({ ok: true, pids: [4242] }); + } finally { + scan.mockRestore(); + } + } + }); +}); + +describe("scanListenPidsForAddress (real scanner)", () => { + test("finds this process on its own bound port and filters other addresses", async () => { + const server = createServer(); + await new Promise((resolve, reject) => { + server.once("error", reject); + server.listen(0, "127.0.0.1", () => resolve()); + }); + try { + const address = server.address(); + if (typeof address === "object" && address) { + const scan = scanListenPidsForAddress(address.port, "127.0.0.1"); + // Missing platform tools must report a failed scan rather than an empty result. + if (scan.ok) expect(scan.pids).toContain(process.pid); + } + } finally { + server.close(); + } + }); + + test("a listener on another loopback address does not serve 127.0.0.1", async () => { + const server = createServer(); + const bound = await new Promise(resolve => { + server.once("error", () => resolve(false)); + server.listen(0, "127.0.0.2", () => resolve(true)); + }); + if (!bound) return; + try { + const address = server.address(); + if (typeof address === "object" && address) { + const scan = scanListenPidsForAddress(address.port, "127.0.0.1"); + if (scan.ok) expect(scan.pids).not.toContain(process.pid); + const wide = scanListenPidsForAddress(address.port, "0.0.0.0"); + if (wide.ok) expect(wide.pids).toContain(process.pid); + } + } finally { + server.close(); + } + }); +}); + describe("parseTcpQuadsForLocalPort / IPv6", () => { test("collects every TCP row on the local port including non-LISTEN states", () => { const output = [ From b697756cdb3bacee14d9bac614a352475d92f739 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 15:27:59 +0900 Subject: [PATCH 33/75] fix(update): avoid PATH lookup for systemd-run (#6037) Carried from #6037 into merge train round 3. Resolved the src/update/job.ts import conflict with dev by keeping both imports. Co-authored-by: Epinephrine --- src/server/management/config-routes.ts | 11 +- src/update/job.ts | 4 +- src/update/worker-launch.ts | 163 ++++++++++++++-- structure/ops/service-and-sidecars.md | 2 +- tests/update/update-worker-launch.test.ts | 221 +++++++++++++++++++++- 5 files changed, 381 insertions(+), 20 deletions(-) diff --git a/src/server/management/config-routes.ts b/src/server/management/config-routes.ts index adf7c639253..45cfee41f76 100644 --- a/src/server/management/config-routes.ts +++ b/src/server/management/config-routes.ts @@ -772,7 +772,7 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise checked, + spawnWorkerFn: (jobId, runChannel, runRestart) => + spawnGuiUpdateWorker(jobId, runChannel, runRestart, { resolveSystemdRun: () => systemdRun }), }) }); } catch (err) { if (err instanceof UpdateJobError) { diff --git a/src/update/job.ts b/src/update/job.ts index fd160f8d568..8f90b0cad32 100644 --- a/src/update/job.ts +++ b/src/update/job.ts @@ -62,6 +62,7 @@ import { } from "./npm-cache-preflight.mjs"; import { guiUpdateWorkerCommand } from "./worker-launch"; import { withoutSiblingMarker } from "../codex/sibling-start"; +import type { WorkerLaunchContext } from "./worker-launch"; const RELEASE_NOTES_URL = "https://github.com/lidge-jun/opencodex/releases/latest"; const UPDATE_JOB_FILENAME = "update-job.json"; @@ -568,6 +569,7 @@ export function spawnGuiUpdateWorker( jobId: string, channel: Channel, restart: boolean, + context: WorkerLaunchContext = {}, ): UpdateWorkerProcess { const args = selfLaunchArgv([ "__gui-update-worker", @@ -576,7 +578,7 @@ export function spawnGuiUpdateWorker( restart ? "restart" : "no-restart", ]); if (process.platform !== "win32") { - const launch = guiUpdateWorkerCommand(process.execPath, args); + const launch = guiUpdateWorkerCommand(process.execPath, args, context); return spawn(launch.command, launch.argv, { detached: true, stdio: "ignore", diff --git a/src/update/worker-launch.ts b/src/update/worker-launch.ts index 89811193e1f..3c964c3b489 100644 --- a/src/update/worker-launch.ts +++ b/src/update/worker-launch.ts @@ -1,4 +1,6 @@ -import { spawnSync } from "node:child_process"; +import { spawn, spawnSync } from "node:child_process"; +import { accessSync, constants, realpathSync, statSync } from "node:fs"; +import { dirname, isAbsolute } from "node:path"; /** * How to launch the dashboard update worker on POSIX. @@ -16,19 +18,157 @@ export const SYSTEMD_SCOPE_ARGS = ["--user", "--scope", "--quiet", "--collect", export interface WorkerLaunchContext { platform?: NodeJS.Platform; env?: NodeJS.ProcessEnv; - hasSystemdRun?: () => boolean; + resolveSystemdRun?: () => string | undefined; } -let systemdRunProbe: boolean | undefined; +// Absolute install paths only — PATH is never consulted, so a caller-controlled entry cannot +// redirect the launch. `/usr/local/bin` is where systemd lands when built or stowed outside the +// distro layout, and `/run/current-system/sw/bin` is the NixOS layout, where the binary lives +// nowhere else even though the user bus works. A candidate only counts when the binary and its +// directory are root-owned and not group/world-writable, so a lower-trust local actor cannot +// plant the launcher the scope probe execs. +const TRUSTED_SYSTEMD_RUN_PATHS = [ + "/usr/bin/systemd-run", "/bin/systemd-run", "/usr/local/bin/systemd-run", + "/run/current-system/sw/bin/systemd-run", +] as const; -function probeSystemdRun(): boolean { +export interface SystemdRunHooks { + isExecutableFile: (path: string) => boolean; + probeScope: (path: string) => boolean; + /** Async variant of probeScope; resolveSystemdRunAsync prefers it when present. */ + probeScopeAsync?: (path: string) => Promise; +} + +const GROUP_OR_WORLD_WRITE = 0o022; + +// stat (follow) rather than lstat: a root-owned symlink to a user-writable directory must fail +// on the target's mode, not pass on the symlink's (mirrors isTrustedSystemPath in +// src/codex/desktop-app/linux.ts). +export interface SystemdRunTrustDeps { + /** Test seam: canonicalizes the candidate before its substitution chain is checked. */ + realpathSync?: (path: string) => string; + /** Test seam: stats a resolved path for ownership and mode. */ + statSync?: (path: string) => { isFile(): boolean; uid: number; mode: number }; + /** Test seam: checks the candidate's executable bit. */ + accessSync?: (path: string, mode: number) => void; +} + +function rootOnlyWritable(path: string, stat: SystemdRunTrustDeps["statSync"] = statSync): boolean { + try { + const st = stat!(path); + return st.uid === 0 && (st.mode & GROUP_OR_WORLD_WRITE) === 0; + } catch { + return false; + } +} + +// "Executable" here includes trust: the binary and its directory must be root-owned and not +// group/world-writable. /usr/local/bin is group-writable on some systems, and a planted or +// replaced systemd-run there would be exec'd by the scope probe under the service account; +// the fallback is the plain detached spawn, so nothing breaks when it is skipped. +// Exported for unit tests. +export function isTrustedSystemdRunFile(path: string, deps: SystemdRunTrustDeps = {}): boolean { + try { + if (!isAbsolute(path)) return false; + (deps.accessSync ?? accessSync)(path, constants.X_OK); + // The lexical path may be a symlink. Checking the link's own parent only proves + // the *entry* is pinned; the file it resolves to — and every ancestor able to + // substitute that resolved file — is what the scope probe will actually exec. + const realpath = deps.realpathSync ?? realpathSync; + const resolved = realpath(path); + const stat = deps.statSync ?? statSync; + const st = stat(resolved); + if (!(st.isFile() && st.uid === 0 && (st.mode & GROUP_OR_WORLD_WRITE) === 0)) { + return false; + } + for (const start of [dirname(path), dirname(resolved)]) { + for (let dir = start, previous = ""; dir !== previous; previous = dir, dir = dirname(dir)) { + if (!rootOnlyWritable(dir, stat)) return false; + } + } + return true; + } catch { + return false; + } +} + +/** Scope discovery gets only user-bus identity, never inherited management credentials. */ +function scopeProbeEnvironment(): NodeJS.ProcessEnv { + const env: NodeJS.ProcessEnv = { PATH: "/usr/bin:/bin" }; + for (const name of ["HOME", "USER", "LOGNAME", "XDG_RUNTIME_DIR", "DBUS_SESSION_BUS_ADDRESS"]) { + if (process.env[name] !== undefined) env[name] = process.env[name]; + } + return env; +} + +const systemdRunHooks: SystemdRunHooks = { + isExecutableFile: isTrustedSystemdRunFile, + // Run a real scope with the same absolute binary as its harmless version payload. + // Probing only the outer --version would not verify the user bus. + probeScope: path => { + const probe = spawnSync(path, [...SYSTEMD_SCOPE_ARGS, path, "--version"], { stdio: "ignore", timeout: 5_000, env: scopeProbeEnvironment() }); + return !probe.error && probe.status === 0; + }, + probeScopeAsync: path => new Promise(resolve => { + const probe = spawn(path, [...SYSTEMD_SCOPE_ARGS, path, "--version"], { stdio: "ignore", env: scopeProbeEnvironment() }); + probe.unref(); + const timer = setTimeout(() => { + try { probe.kill("SIGKILL"); } catch { /* failed termination is not a successful probe */ } + resolve(false); + }, 5_000); + timer.unref(); + probe.once("error", () => { clearTimeout(timer); resolve(false); }); + probe.once("close", code => { clearTimeout(timer); resolve(code === 0); }); + }), +}; + +let systemdRunProbe: string | null | undefined; +let systemdRunProbePending: Promise | undefined; + +export function resolveSystemdRun(hooks: SystemdRunHooks = systemdRunHooks): string | undefined { if (systemdRunProbe === undefined) { - // Run a real no-op scope rather than `--version`: a present binary without a reachable user - // bus would otherwise pass the probe and then fail to start the worker at all. - const probe = spawnSync("systemd-run", [...SYSTEMD_SCOPE_ARGS, "true"], { stdio: "ignore", timeout: 5_000 }); - systemdRunProbe = !probe.error && probe.status === 0; + systemdRunProbe = null; + for (const command of TRUSTED_SYSTEMD_RUN_PATHS) { + if (!hooks.isExecutableFile(command)) continue; + if (hooks.probeScope(command)) { + systemdRunProbe = command; + break; + } + } } - return systemdRunProbe; + return systemdRunProbe ?? undefined; +} + +/** + * Management-request path variant. The synchronous resolver blocks the shared + * event loop for up to four sequential five-second scope probes on first use; + * the dashboard update route awaits this instead, so probing overlaps other + * requests. Concurrent first callers share one probe pass. + */ +export async function resolveSystemdRunAsync(hooks: SystemdRunHooks = systemdRunHooks): Promise { + if (systemdRunProbe !== undefined) return systemdRunProbe ?? undefined; + if (!systemdRunProbePending) { + systemdRunProbePending = (async () => { + const probeScope = hooks.probeScopeAsync ?? (async (path: string) => hooks.probeScope(path)); + for (const command of TRUSTED_SYSTEMD_RUN_PATHS) { + if (!hooks.isExecutableFile(command)) continue; + if (await probeScope(command)) { + return command; + } + } + return null; + })(); + } + const found = await systemdRunProbePending; + // Honor a cache the sync resolver may have filled while the probe ran — the + // older observation wins so every caller converges on one launcher. + if (systemdRunProbe === undefined) systemdRunProbe = found; + return systemdRunProbe ?? undefined; +} + +export function resetSystemdRunProbeForTests(): void { + systemdRunProbe = undefined; + systemdRunProbePending = undefined; } export function guiUpdateWorkerCommand( @@ -39,8 +179,9 @@ export function guiUpdateWorkerCommand( const platform = context.platform ?? process.platform; const env = context.env ?? process.env; const underSystemd = platform === "linux" && Boolean(env.INVOCATION_ID); - if (underSystemd && (context.hasSystemdRun ?? probeSystemdRun)()) { - return { command: "systemd-run", argv: [...SYSTEMD_SCOPE_ARGS, execPath, ...args] }; + const systemdRun = underSystemd ? (context.resolveSystemdRun ?? resolveSystemdRun)() : undefined; + if (systemdRun) { + return { command: systemdRun, argv: [...SYSTEMD_SCOPE_ARGS, execPath, ...args] }; } return { command: execPath, argv: [...args] }; } diff --git a/structure/ops/service-and-sidecars.md b/structure/ops/service-and-sidecars.md index d8702d82b46..8a2d164a7f6 100644 --- a/structure/ops/service-and-sidecars.md +++ b/structure/ops/service-and-sidecars.md @@ -332,4 +332,4 @@ src/update/async-check.ts uses the existing owner-bound registry target with a b The desktop badge snapshot in src/update/desktop-badge.ts is process-local display state keyed by a Tauri session id. A 60-second shell heartbeat renews receipt time; entries expire after 180 seconds and the store retains at most 32 sessions. It is separate from the package version cache and from the updater job/ownership transaction. A proxy restart reports unknown until a bound desktop shell republishes; no update installation can be authorized by this snapshot. -On Linux, a dashboard update worker started from the systemd user service is launched through `systemd-run --user --scope --quiet --collect` (`src/update/worker-launch.ts`), so it leaves the service cgroup before the updater stops `opencodex-proxy.service`; the default `KillMode=control-group` otherwise kills it with the proxy (#5750). The path applies only when `INVOCATION_ID` is set and a no-op scope probe succeeds; every other case keeps the plain detached spawn. `--scope` moves `systemd-run` itself into the scope and then execs the worker, so the recorded PID is the worker's (`tests/update/update-worker-launch.test.ts`). +On Linux, a dashboard update worker started from the systemd user service is launched through an executable regular file at a trusted absolute path — `/usr/bin/systemd-run`, `/bin/systemd-run`, `/usr/local/bin/systemd-run` (local installs), or `/run/current-system/sw/bin/systemd-run` (the NixOS layout) — with `--user --scope --quiet --collect` (`src/update/worker-launch.ts`), so it leaves the service cgroup before the updater stops `opencodex-proxy.service`; the default `KillMode=control-group` otherwise kills it with the proxy (#5750). The inherited `PATH` is never searched, and each candidate's resolved target — plus every ancestor directory able to substitute it — must be root-owned and not group/world-writable: a trusted-path symlink into a user-replaceable directory is skipped, as is a group-writable `/usr/local/bin`, rather than exec'd under the service account. Candidates are tried in order and a path whose no-op scope probe fails falls through to the next trusted path; the probe applies only when `INVOCATION_ID` is set, and every other case keeps the plain detached spawn. The management route resolves the launcher with `resolveSystemdRunAsync` before spawning, so first-request probing overlaps other work instead of blocking the event loop for up to twenty seconds. `--scope` moves `systemd-run` itself into the scope and then execs the worker, so the recorded PID is the worker's (`tests/update/update-worker-launch.test.ts`). diff --git a/tests/update/update-worker-launch.test.ts b/tests/update/update-worker-launch.test.ts index 4132df54912..661c4fea581 100644 --- a/tests/update/update-worker-launch.test.ts +++ b/tests/update/update-worker-launch.test.ts @@ -1,5 +1,12 @@ import { describe, expect, test } from "bun:test"; -import { guiUpdateWorkerCommand, SYSTEMD_SCOPE_ARGS } from "../../src/update/worker-launch"; +import { chmodSync, mkdtempSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { + guiUpdateWorkerCommand, isTrustedSystemdRunFile, resolveSystemdRun, resetSystemdRunProbeForTests, + resolveSystemdRunAsync, SYSTEMD_SCOPE_ARGS, +} from "../../src/update/worker-launch"; +import { removeTreeWithRetry } from "../helpers/remove-tree"; // #5750: a worker spawned by the systemd user service must leave the service cgroup before the // updater stops that service, or systemd kills it along with the proxy. @@ -8,23 +15,225 @@ describe("dashboard update worker launch", () => { test("a systemd-started Linux proxy launches the worker in its own scope", () => { const launch = guiUpdateWorkerCommand("/usr/bin/bun", args, { - platform: "linux", env: { INVOCATION_ID: "abc" }, hasSystemdRun: () => true, + platform: "linux", env: { INVOCATION_ID: "abc", PATH: "/tmp/attacker:/usr/bin" }, + resolveSystemdRun: () => "/usr/bin/systemd-run", + }); + expect(launch).toEqual({ + command: "/usr/bin/systemd-run", argv: [...SYSTEMD_SCOPE_ARGS, "/usr/bin/bun", ...args], }); - expect(launch).toEqual({ command: "systemd-run", argv: [...SYSTEMD_SCOPE_ARGS, "/usr/bin/bun", ...args] }); }); test("without systemd-run, outside systemd, or off Linux the spawn is unchanged", () => { const plain = { command: "/usr/bin/bun", argv: args }; expect(guiUpdateWorkerCommand("/usr/bin/bun", args, { - platform: "linux", env: { INVOCATION_ID: "abc" }, hasSystemdRun: () => false, + platform: "linux", env: { INVOCATION_ID: "abc" }, resolveSystemdRun: () => undefined, })).toEqual(plain); let probed = false; expect(guiUpdateWorkerCommand("/usr/bin/bun", args, { - platform: "linux", env: {}, hasSystemdRun: () => { probed = true; return true; }, + platform: "linux", env: {}, resolveSystemdRun: () => { probed = true; return "/usr/bin/systemd-run"; }, })).toEqual(plain); expect(probed).toBe(false); expect(guiUpdateWorkerCommand("/usr/bin/bun", args, { - platform: "darwin", env: { INVOCATION_ID: "abc" }, hasSystemdRun: () => true, + platform: "darwin", env: { INVOCATION_ID: "abc" }, resolveSystemdRun: () => "/usr/bin/systemd-run", })).toEqual(plain); }); }); + +// The real resolver — not the context seam — must be the thing under test: PATH must stay +// unconsulted, only the trusted absolute candidates may be probed, and a failed probe must +// fall through rather than settle for the plain in-cgroup spawn. +describe("trusted systemd-run discovery", () => { + test("walks only the trusted candidates and ignores PATH", () => { + resetSystemdRunProbeForTests(); + const seen: string[] = []; + const found = resolveSystemdRun({ + isExecutableFile: path => { seen.push(path); return path === "/run/current-system/sw/bin/systemd-run"; }, + probeScope: () => true, + }); + expect(found).toBe("/run/current-system/sw/bin/systemd-run"); + expect(seen).toEqual([ + "/usr/bin/systemd-run", "/bin/systemd-run", "/usr/local/bin/systemd-run", + "/run/current-system/sw/bin/systemd-run", + ]); + expect(seen.every(path => path.startsWith("/"))).toBe(true); + }); + + test("a failed scope probe falls through to the next candidate", () => { + resetSystemdRunProbeForTests(); + const found = resolveSystemdRun({ + isExecutableFile: () => true, + probeScope: path => path !== "/usr/bin/systemd-run", + }); + expect(found).toBe("/bin/systemd-run"); + }); + + test("the probe is cached and reports undefined when nothing qualifies", () => { + resetSystemdRunProbeForTests(); + let calls = 0; + const hooks = { + isExecutableFile: () => { calls++; return false; }, + probeScope: () => { throw new Error("must not run"); }, + }; + expect(resolveSystemdRun(hooks)).toBeUndefined(); + expect(resolveSystemdRun(hooks)).toBeUndefined(); + expect(calls).toBe(4); + resetSystemdRunProbeForTests(); + }); + + test("resolveSystemdRunAsync shares one probe pass across concurrent first callers", async () => { + resetSystemdRunProbeForTests(); + let probes = 0; + const hooks = { + isExecutableFile: () => true, + probeScope: () => { throw new Error("sync probe must not run on the request path"); }, + probeScopeAsync: async (path: string) => { + probes++; + await new Promise(resolve => setTimeout(resolve, 5)); + return path === "/bin/systemd-run"; + }, + }; + const [first, second, third] = await Promise.all([ + resolveSystemdRunAsync(hooks), + resolveSystemdRunAsync(hooks), + resolveSystemdRunAsync(hooks), + ]); + expect(first).toBe("/bin/systemd-run"); + expect(second).toBe("/bin/systemd-run"); + expect(third).toBe("/bin/systemd-run"); + expect(probes).toBe(2); + // The resolved value is now cached: the sync resolver agrees without probing again. + expect(resolveSystemdRun(hooks)).toBe("/bin/systemd-run"); + expect(probes).toBe(2); + resetSystemdRunProbeForTests(); + }); +}); + +// The default trust check must run against the real filesystem, not a stubbed seam. uid/mode +// semantics are POSIX-only — on Windows statSync reports uid 0 and chmod is a no-op — and only +// a root-run suite can create a uid-0 fixture, so each case is gated on what the test user can +// actually arrange. +describe("isTrustedSystemdRunFile (real filesystem)", () => { + const posix = process.platform !== "win32"; + const itPosix = posix ? test : test.skip; + const getuid = (process as { getuid?: () => number }).getuid?.bind(process); + const itNonRoot = posix && getuid?.() !== 0 ? test : test.skip; + const itRoot = posix && getuid?.() === 0 ? test : test.skip; + + function fixture(): { dir: string; file: string; cleanup: () => void } { + const dir = mkdtempSync(join(tmpdir(), "ocx-systemd-run-trust-")); + const file = join(dir, "systemd-run"); + writeFileSync(file, "#!/bin/sh\nexit 0\n"); + chmodSync(file, 0o755); + return { dir, file, cleanup: () => removeTreeWithRetry(dir) }; + } + + itNonRoot("rejects an executable owned by the test user rather than root", () => { + const { file, cleanup } = fixture(); + try { expect(isTrustedSystemdRunFile(file)).toBe(false); } finally { cleanup(); } + }); + + itNonRoot("rejects non-executable and missing paths", () => { + const { dir, file, cleanup } = fixture(); + try { + chmodSync(file, 0o644); + expect(isTrustedSystemdRunFile(file)).toBe(false); + expect(isTrustedSystemdRunFile(join(dir, "absent"))).toBe(false); + expect(isTrustedSystemdRunFile(dir)).toBe(false); + } finally { cleanup(); } + }); + + itRoot("rejects a root-owned file inside a group/world-writable directory", () => { + const { dir, file, cleanup } = fixture(); + try { + chmodSync(dir, 0o777); + expect(isTrustedSystemdRunFile(file)).toBe(false); + } finally { + chmodSync(dir, 0o700); + cleanup(); + } + }); + + itRoot("accepts a root-owned executable in a root-only-writable directory", () => { + const { dir, file, cleanup } = fixture(); + try { + chmodSync(dir, 0o755); + expect(isTrustedSystemdRunFile(file)).toBe(true); + } finally { cleanup(); } + }); +}); + +/* + * A trusted-path symlink is only as strong as the file it resolves to and the + * directories able to substitute that file. The link's own parent being + * root-only is not enough — these run against stub seams so the substitution + * chain is exercised without needing a uid-0 fixture on disk. + */ +describe("isTrustedSystemdRunFile (resolved substitution chain)", () => { + const fileStat = (mode: number, uid = 0) => ({ isFile: () => true, isDirectory: () => false, uid, mode }); + const dirStat = (mode: number, uid = 0) => ({ isFile: () => false, isDirectory: () => true, uid, mode }); + const trustedDeps = { + accessSync: () => {}, + statSync: (path: string) => dirStat(0o755), + realpathSync: (path: string) => path, + }; + + test("rejects a trusted-dir symlink whose resolved target can be substituted", () => { + // /usr/bin/systemd-run -> /home/user/bin/systemd-run: the file itself is + // root-owned and mode-pinned, but /home/user/bin is user-writable, so the + // user can replace it outright. + const deps = { + ...trustedDeps, + realpathSync: () => "/home/user/bin/systemd-run", + statSync: (path: string) => + path === "/home/user/bin/systemd-run" ? fileStat(0o755) + : path === "/home/user/bin" ? dirStat(0o775) + : dirStat(0o755), + }; + expect(isTrustedSystemdRunFile("/usr/bin/systemd-run", deps)).toBe(false); + }); + + test("rejects when any resolved ancestor can be substituted, not just the parent", () => { + // Target dir is pinned, but /opt/vendor is world-writable: swapping + // /opt/vendor/tools there substitutes the binary below it. + const deps = { + ...trustedDeps, + realpathSync: () => "/opt/vendor/tools/systemd-run", + statSync: (path: string) => + path === "/opt/vendor/tools/systemd-run" ? fileStat(0o755) + : path === "/opt/vendor" ? dirStat(0o777) + : dirStat(0o755), + }; + expect(isTrustedSystemdRunFile("/usr/bin/systemd-run", deps)).toBe(false); + }); + + test("accepts a resolved chain that is root-owned and pinned end to end", () => { + const deps = { + ...trustedDeps, + realpathSync: () => "/usr/lib/systemd/systemd-run", + statSync: (path: string) => + path === "/usr/lib/systemd/systemd-run" ? fileStat(0o755) : dirStat(0o755), + }; + expect(isTrustedSystemdRunFile("/usr/bin/systemd-run", deps)).toBe(true); + }); + + test("rejects a non-root resolved target even inside a pinned chain", () => { + const deps = { + ...trustedDeps, + realpathSync: () => "/usr/lib/systemd/systemd-run", + statSync: (path: string) => + path === "/usr/lib/systemd/systemd-run" ? fileStat(0o755, 1000) : dirStat(0o755), + }; + expect(isTrustedSystemdRunFile("/usr/bin/systemd-run", deps)).toBe(false); + }); +}); + +test("launcher trust also checks lexical ancestors of a canonical system target", () => { + const candidate = "/usr/local/bin/systemd-run"; + const target = "/nix/store/systemd/bin/systemd-run"; + const deps = (bad: string | undefined) => ({ realpathSync: () => target, accessSync: () => {}, + statSync: (path: string) => ({ isFile: () => path === target, uid: 0, mode: path === bad ? 0o777 : 0o755 }) }); + expect(isTrustedSystemdRunFile(candidate, deps(undefined))).toBe(true); + expect(isTrustedSystemdRunFile(candidate, deps("/usr/local"))).toBe(false); + expect(isTrustedSystemdRunFile(candidate, deps("/nix/store"))).toBe(false); + expect(isTrustedSystemdRunFile("relative/systemd-run", deps(undefined))).toBe(false); +}); From 8e08f26f8670d747fc63fb36621c9981c8d1f230 Mon Sep 17 00:00:00 2001 From: Terry Tan Date: Sun, 27 Sep 2026 15:28:04 +0900 Subject: [PATCH 34/75] fix(codex): activate quota from saved deadlines (#6020) Carried from #6020 into merge train round 3. Co-authored-by: Terry Tan --- .../docs/getting-started/how-it-works.mdx | 9 +- .../zh-cn/getting-started/how-it-works.mdx | 6 +- src/codex/quota-auto-refresh-state.ts | 6 +- src/codex/quota-auto-refresh.ts | 37 +++++-- structure/catalog.md | 8 +- structure/codex-home.md | 2 + structure/config.md | 6 +- structure/gui-and-management-api.md | 2 + structure/ops/docs-and-release.md | 2 + structure/providers/openai-tiers.md | 13 ++- structure/runtime.md | 6 +- structure/subagents.md | 2 + ...-quota-auto-refresh-main-admission.test.ts | 5 +- .../codex-quota-auto-refresh.test.ts | 104 ++++++++++++++++++ 14 files changed, 179 insertions(+), 29 deletions(-) diff --git a/docs-site/src/content/docs/getting-started/how-it-works.mdx b/docs-site/src/content/docs/getting-started/how-it-works.mdx index c75ffed90e4..76b489f2812 100644 --- a/docs-site/src/content/docs/getting-started/how-it-works.mdx +++ b/docs-site/src/content/docs/getting-started/how-it-works.mdx @@ -51,8 +51,13 @@ account before the request is forwarded upstream. The rule is intentionally spli coalesces simultaneous windows into one request, and durably persists both reset timestamps to prevent duplicate work after restarts. Paused accounts and accounts requiring reauthentication are skipped. Activation captures successful response quota headers; - opted-in idle accounts also refresh stale quota metadata at most once every five minutes, - without needing an open dashboard. Observed reset boundaries are retained across restarts + known reset times are checked locally each minute without periodic quota queries, even when + the cached usage is old or the proxy restarts. Only missing reset times need a metadata query + after the five-minute freshness guard. Unresolved discovery and failed activations retry after + 5, 10, 20, 40, then at most every 60 minutes; these retry delays reset on proxy restart. + Successful response headers seed the next window without an extra query when available. + Dashboard refreshes and optional reset-notification polling remain independent. + Observed reset boundaries are retained across restarts until completed, so a moving idle-window timestamp cannot erase a pending activation. Metadata refresh uses the existing bounded authentication recovery; an inference 401 marks the rejected credential for reauthentication instead of repeatedly spending retries on it. diff --git a/docs-site/src/content/docs/zh-cn/getting-started/how-it-works.mdx b/docs-site/src/content/docs/zh-cn/getting-started/how-it-works.mdx index d234dd1ff5d..dbd75768d0a 100644 --- a/docs-site/src/content/docs/zh-cn/getting-started/how-it-works.mdx +++ b/docs-site/src/content/docs/zh-cn/getting-started/how-it-works.mdx @@ -38,7 +38,11 @@ Codex 使用 OpenAI **Responses API**。opencodex 接收通过 HTTP 与 Server-S 已报告的 5 小时及每周窗口;新添加账号不会自动启用。在 Pool 模式下,窗口到期后会通过对应账号 发送最小化、不保存的请求,并消耗少量额度;同时到期的窗口合并为一次请求。暂停、需要重新认证 的账号会被跳过,主账号硬锁限制也会得到遵守。成功响应的额度头会更新缓存;已启用且符合条件的 - 空闲账号还会每隔至少 5 分钟刷新过期的额度元数据,无需保持仪表盘打开。已观察到的到期时间会保留 + 空闲账号已有重置时间时,每分钟仅在本地检查是否到期,不会因缓存过期或代理重启而定期查询额度。 + 只有缺少重置时间时,才在五分钟新鲜度保护后补查。持续缺失信息或激活失败时,重试间隔依次为 + 5、10、20、40、60 分钟,并以 60 分钟封顶;重试间隔在代理重启后重新计算。 + 成功响应头提供下一轮时间时无需额外查询。仪表盘刷新及可选的重置通知轮询仍独立运行。 + 已观察到的到期时间会保留 至激活完成,重启或后续查询的时间变化不会丢失待处理窗口。元数据查询复用现有的有次数限制的认证 恢复逻辑;推理请求返回 401 时,被拒绝的凭据会标记为需要重新认证。失败日志仅记录不透明账号标签 和安全的状态原因。该功能独立于为传入请求选择账号的路由逻辑。 diff --git a/src/codex/quota-auto-refresh-state.ts b/src/codex/quota-auto-refresh-state.ts index 75ebe0db64e..149d9b97fc4 100644 --- a/src/codex/quota-auto-refresh-state.ts +++ b/src/codex/quota-auto-refresh-state.ts @@ -2,10 +2,12 @@ /** Completed/due markers use epoch milliseconds; persisted legacy markers may use seconds. */ export type CodexQuotaAutoRefreshWindows = { fiveHour?: number; weekly?: number }; +export type CodexQuotaRetry = { after: number; delay: number }; + export const completedByAccount = new Map(); -export const retryAfterByAccount = new Map(); +export const retryAfterByAccount = new Map(); export const scheduledByAccount = new Map(); -export const quotaRefreshAfterByAccount = new Map(); +export const quotaRefreshAfterByAccount = new Map(); /** Drop every activation record when its account is removed. */ export function forgetCodexQuotaAutoRefreshAccount(accountId: string): void { diff --git a/src/codex/quota-auto-refresh.ts b/src/codex/quota-auto-refresh.ts index cf21a46e169..53b8eb81cda 100644 --- a/src/codex/quota-auto-refresh.ts +++ b/src/codex/quota-auto-refresh.ts @@ -21,13 +21,14 @@ import { CodexWarmupError, codexWarmupFailureReason, warmCodexAccount } from "./ import { completedByAccount, retryAfterByAccount, scheduledByAccount, quotaRefreshAfterByAccount, resetCodexQuotaAutoRefreshStateForTests, - type CodexQuotaAutoRefreshWindows, + type CodexQuotaAutoRefreshWindows, type CodexQuotaRetry, } from "./quota-auto-refresh-state"; export type { CodexQuotaAutoRefreshWindows } from "./quota-auto-refresh-state"; export { forgetCodexQuotaAutoRefreshAccount } from "./quota-auto-refresh-state"; export const FIVE_HOUR_WINDOW_SECONDS = 5 * 60 * 60; const RETRY_MS = 5 * 60_000; +const MAX_RETRY_MS = 60 * 60_000; const CONCURRENCY = 4; export interface CodexQuotaAutoRefreshStatus { @@ -51,6 +52,21 @@ export interface CodexQuotaAutoRefreshRunDeps { let inFlight: Promise | null = null; +/** Back off unsuccessful discovery/activation without adding another timer. */ +function deferRetry(retries: Map, accountId: string, now: number): number { + const delay = Math.min((retries.get(accountId)?.delay ?? RETRY_MS / 2) * 2, MAX_RETRY_MS); + retries.set(accountId, { after: now + delay, delay }); + return delay; +} + +/** Every enabled window needs a retained, uncompleted deadline, not fresh usage percentages. */ +function hasScheduledWindows(config: OcxConfig, accountId: string): boolean { + const setting = config.codexQuotaAutoRefresh?.[accountId]; + const scheduled = scheduledByAccount.get(accountId); + return (!setting?.fiveHour || scheduled?.fiveHour !== undefined) + && (!setting?.weekly || scheduled?.weekly !== undefined); +} + /** Report upstream window availability separately from persisted spending intent. */ export function codexQuotaAutoRefreshStatus( config: OcxConfig, @@ -309,14 +325,19 @@ export async function runCodexQuotaAutoRefresh( // Capture before WHAM can move an idle window's reset into the future. rememberWindows(config, accountId, quotaFor(accountId)); const quota = quotaFor(accountId); - if ((!quota || now - quota.updatedAt >= RETRY_MS) - && (quotaRefreshAfterByAccount.get(accountId) ?? 0) <= now) { - quotaRefreshAfterByAccount.set(accountId, now + RETRY_MS); - try { await refresh(config, accountId); } catch { /* Retry metadata at the bounded cadence. */ } + // A known deadline remains actionable even when its usage snapshot is old. + // Only discover missing windows; never poll merely to keep percentages fresh. + if (hasScheduledWindows(config, accountId)) { + quotaRefreshAfterByAccount.delete(accountId); + } else if ((!quota || now - quota.updatedAt >= RETRY_MS) + && (quotaRefreshAfterByAccount.get(accountId)?.after ?? 0) <= now) { + deferRetry(quotaRefreshAfterByAccount, accountId, now); + try { await refresh(config, accountId); } catch { /* Retry missing metadata with backoff. */ } } if (!eligible(accountId)) return; rememberWindows(config, accountId, quotaFor(accountId)); - if ((retryAfterByAccount.get(accountId) ?? 0) > now) return; + if (hasScheduledWindows(config, accountId)) quotaRefreshAfterByAccount.delete(accountId); + if ((retryAfterByAccount.get(accountId)?.after ?? 0) > now) return; const windows = dueCodexQuotaAutoRefreshWindows(config, accountId, quotaFor(accountId), now); if (!windows) return; try { @@ -327,11 +348,11 @@ export async function runCodexQuotaAutoRefresh( persist(config, accountId, completed); rememberWindows(config, accountId, quotaFor(accountId)); } catch (error) { - retryAfterByAccount.set(accountId, now + RETRY_MS); + const delay = deferRetry(retryAfterByAccount, accountId, now); const account = config.codexAccounts?.find(candidate => candidate.id === accountId); const label = account ? codexAccountLogLabel(account) : "main"; console.warn(`[codex-quota-auto-refresh] ${label}: ${codexWarmupFailureReason(error)}; ${ - isAccountNeedsReauth(accountId) ? "reauthentication required" : "retry in five minutes" + isAccountNeedsReauth(accountId) ? "reauthentication required" : `retry in ${delay / 60_000} minutes` }`); } })); diff --git a/structure/catalog.md b/structure/catalog.md index d0d054768b5..0c79ecbc20c 100644 --- a/structure/catalog.md +++ b/structure/catalog.md @@ -1,5 +1,7 @@ # Model Catalog +Activation-owned metadata discovery no longer refreshes known deadlines merely because quota snapshots age. See the [quota activation contract](providers/openai-tiers.md#public-provider-contract). + Native result continuations and function-result injection follow [the mode-specific result and control contract](transports/streaming-health.md#experimental-native-function-result-injection); this surface does not infer upstream support or alter its defaults. Explicit Codex CLI installation observation supplies no selected-runtime proof to catalog discovery or publication. See the [read-only observation contract](runtime.md#explicit-codex-cli-installation-observation). @@ -593,8 +595,6 @@ Subagent account previews and live routing share the [priority failback](provide Startup and explicit catalog synchronization in `src/codex/sync.ts` refresh the optional `src/providers/reasoning-metadata.ts` effort snapshot for supported destinations before catalog -gathering. Each sync waits at most two seconds for a fresh or shared fetch, then continues with -the existing snapshot; the fetch retains its own abort deadline. Routed effort reads in +gathering. Each sync waits at most two seconds for a fresh or shared fetch, then continues with the existing snapshot; the fetch retains its own abort deadline. Routed effort reads in `src/reasoning-effort.ts` use a snapshot immediately and request a best-effort background refresh -only when an existing snapshot answers with an expired ladder. Missing or corrupt snapshots do -not fetch on the request path; catalog sync owns their bootstrap. +only when an existing snapshot answers with an expired ladder. Missing or corrupt snapshots do not fetch on the request path; catalog sync owns their bootstrap. diff --git a/structure/codex-home.md b/structure/codex-home.md index d14ee6bfcdc..b7dfeb0860d 100644 --- a/structure/codex-home.md +++ b/structure/codex-home.md @@ -1,5 +1,7 @@ # Codex Home +Quota activation restores deadlines from OpenCodex settings; its retry backoff remains process-local. See the [quota activation contract](providers/openai-tiers.md#public-provider-contract). + Catalog HTTP acquisition follows the [proxy-routing contract](catalog.md#remote-catalog-http-proxy-routing). A lock in the Codex credential store is governed by [descriptor identity and age](catalog.md#accounts-namespaces-and-pool-rotation), so the mere presence of its filename is neither acquisition nor release authority. Failed path-identity probes leave the lock for stale recovery and preserve the refresh callback outcome. Cooperating lock metadata changes serialize through the existing SQLite mutation transaction; release keeps the descriptor open through identity comparison and any unlink, then closes it. Failed metadata writes remove only a matching owned path after successful coordination; unknown identity, failed probes or unavailable coordination retain the path for stale recovery. Async refresh work holds no metadata transaction. diff --git a/structure/config.md b/structure/config.md index 1207476bc33..3ce49bf31f3 100644 --- a/structure/config.md +++ b/structure/config.md @@ -1,5 +1,7 @@ # Config Surface +Quota activation reuses the existing next-reset fields without adding a polling configuration key. See the [quota activation contract](providers/openai-tiers.md#public-provider-contract). + Native function-result injection follows [the separate opt-in control contract](transports/streaming-health.md#experimental-native-function-result-injection); this surface does not infer upstream support or alter its defaults. Native steering follows [the shared WebSocket contract](transports/streaming-health.md#experimental-native-mid-turn-steering); this surface's defaults remain unchanged. @@ -589,9 +591,7 @@ being treated as a text model by one and an image target by the other. malformed persisted value is off. `src/config/schema/config-schema.ts` degrades a malformed hand edit to absence so an optional monitoring typo cannot discard providers or credentials. The live-write boundary runs `metricsExportConfigError` in `src/config/diagnostics.ts` before the degrading schema, -so wrong types and unknown nested fields are rejected rather than silently saved. Activation is read -when the server process creates its serve options and therefore requires restart; it adds no setting -to the live `/api/settings` mutation surface. +so wrong types and unknown nested fields are rejected rather than silently saved. Activation is read when the server process creates its serve options and therefore requires restart; it adds no setting to the live `/api/settings` mutation surface. `apiSurfaces` and `protocols` on `src/types/config.ts` are parsed by `src/protocols/settings.ts` only; [Protocol Paths](data-planes/protocol-paths.md#settings) owns their schema handling, meaning and the one writer (`PATCH /api/protocols/settings`), including why closing Messages also writes `claudeCode.enabled` through `commitClaudeCodeBlock` (`src/claude/claude-code-block.ts`, the sentinel-stamping block writer every management route uses). diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index 28c65f4e7a9..be59cb022dd 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -1,5 +1,7 @@ # GUI And Management API +Automatic activation retains its existing settings controls; dashboard quota queries remain independent. See the [quota activation contract](providers/openai-tiers.md#public-provider-contract). + The companion settings contract in `src/companion/` persists menu-bar and widget display preferences, while `src/server/management/companion-routes.ts` exposes those settings and the usage timeline assembled by `src/usage/timeline.ts` to local clients. Query, filter-echo and diff --git a/structure/ops/docs-and-release.md b/structure/ops/docs-and-release.md index b2f84c4a57c..e9c860c8b2f 100644 --- a/structure/ops/docs-and-release.md +++ b/structure/ops/docs-and-release.md @@ -1,5 +1,7 @@ # Docs And Release +The activation scheduling contract is covered by `tests/codex-integration/codex-quota-auto-refresh.test.ts`, including restart recovery and bounded retries. See the [quota activation contract](../providers/openai-tiers.md#public-provider-contract). + Automatic package-tree restart holds a releasable data-plane drain until its scheduled service-home check succeeds. A veto releases that fence; a committed shutdown uses the permanent drain latch. diff --git a/structure/providers/openai-tiers.md b/structure/providers/openai-tiers.md index 733cd1aec8c..00187126f80 100644 --- a/structure/providers/openai-tiers.md +++ b/structure/providers/openai-tiers.md @@ -204,12 +204,17 @@ existing minimal non-stored warmup through that exact account once the timestamp field-patches the completed timestamp. The next observed reset boundary is also retained in `nextFiveHourResetAt` / `nextWeeklyResetAt` until completed; later idle-window metadata cannot postpone it. Successful warmups publish quota headers under the captured credential/identity fence. -For opted-in accounts only, stale metadata is refreshed at most once per five minutes through -the existing WHAM recovery path, independently of dashboard traffic or reset notifications. +Known deadlines suppress activation-owned WHAM queries regardless of snapshot age, including +when only persisted deadlines survive a restart. Missing enabled-window deadlines use the existing +WHAM recovery path after the five-minute freshness guard; unresolved discovery backs off from +five minutes to an hour (5, 10, 20, 40, 60 minutes). Passive headers can satisfy discovery without +a query. Completed warmups seed the next deadlines from response headers; missing next-window +headers use the same discovery path. Retry delays are process-local; deadlines remain durable. +Dashboard queries and reset-notification polling are separate owners and retain their behavior. Inference 401s quarantine the rejected credential; failures log an opaque label and safe reason. Paused or reauthentication-required -accounts are skipped, simultaneous 5-hour/weekly resets share one warmup, transient failures retry -after five minutes, and account deletion removes its setting and completion markers. +accounts are skipped, simultaneous 5-hour/weekly resets share one warmup, transient activation failures +back off from five minutes to an hour, and account deletion removes settings and retry/completion state. Main-account hard-lock also gates these billable warmups. A policy/identity skip changes neither completion markers nor retry delay; quota reads remain available. Main refresh completes before shared credential ownership, then prepared credentials and restrictions are rechecked. Lifecycle diff --git a/structure/runtime.md b/structure/runtime.md index 3fd46e1ca2d..98ddd7ccd46 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -1,5 +1,7 @@ # Runtime +The minute sweep checks persisted activation deadlines locally; only missing deadlines trigger metadata discovery. See the [quota activation contract](providers/openai-tiers.md#public-provider-contract). + ## Resolved static model policy `src/router.ts` attaches one frozen `ResolvedModelPolicy` to every `RouteResult`. Policy/combo @@ -589,9 +591,7 @@ an unreadable current record is unknown, and a valid address is probed even when PID is gone. Lease delegation is passed only to stop and recovery children, never package manager children. A replacement refusal passes through owner-aware recovery: only the same CLI owner revives the stopped runtime; foreign ownership stays transferred and unknown ownership -remains a reported recovery requirement. Dashboard restart delegates the lease token to its repair child. Direct -start holds the same lease through bind plus PID and runtime-address publication. If listener -rollback cannot prove the socket closed, the process retains its lease until exit. +remains a reported recovery requirement. Dashboard restart delegates the lease token to its repair child. Direct start holds the same lease through bind plus PID and runtime-address publication. If listener rollback cannot prove the socket closed, the process retains its lease until exit. The registration is never deleted; `ocx service install` releases the marker only after the registration succeeds. diff --git a/structure/subagents.md b/structure/subagents.md index 3f6370e1fe5..40ec975f6bc 100644 --- a/structure/subagents.md +++ b/structure/subagents.md @@ -1,5 +1,7 @@ # Subagents And Multi-Agent Surface +Subagent quota priming remains separate from automatic activation scheduling. See the [quota activation contract](providers/openai-tiers.md#public-provider-contract). + Native result continuations and function-result injection follow [the mode-specific result and control contract](transports/streaming-health.md#experimental-native-function-result-injection); this surface does not infer upstream support or alter its defaults. Explicit Codex CLI installation observation does not attest the runtime used by a subagent or change agent selection. See the [read-only observation contract](runtime.md#explicit-codex-cli-installation-observation). diff --git a/tests/codex-integration/codex-quota-auto-refresh-main-admission.test.ts b/tests/codex-integration/codex-quota-auto-refresh-main-admission.test.ts index 204436b3a87..946e23b56f3 100644 --- a/tests/codex-integration/codex-quota-auto-refresh-main-admission.test.ts +++ b/tests/codex-integration/codex-quota-auto-refresh-main-admission.test.ts @@ -138,12 +138,13 @@ afterEach(async () => { }); describe("quota auto-refresh native-main admission", () => { - test("stale metadata prepares an expired main token before WHAM and activation", async () => { + test.each([false, true])("expired main token is prepared before activation (missing deadline: %s)", async missingDeadline => { const cfg = config(); writeMain(bearer(true)); const cached = getAccountQuota(MAIN); if (!cached) throw new Error("Expected cached main quota"); cached.updatedAt = now - 300_000; + if (missingDeadline) delete cached.shortResetAt; const fresh = bearer(); const calls = installFetch(async (url, init) => { if (url === tokenUrl) { @@ -157,7 +158,7 @@ describe("quota auto-refresh native-main admission", () => { return completedResponse(); }); await runCodexQuotaAutoRefresh(cfg, now, { persistCompleted: recordMarkers }); - expect(calls).toEqual([tokenUrl, whamUrl, responsesUrl]); + expect(calls).toEqual(missingDeadline ? [tokenUrl, whamUrl, responsesUrl] : [tokenUrl, responsesUrl]); expect(isAccountNeedsReauth(MAIN)).toBe(false); expect(cfg.codexQuotaAutoRefresh?.[MAIN]?.lastFiveHourResetAt).toBe(RESET_MILLISECONDS); expect(getNativeMainProfileRequestCount()).toBe(0); diff --git a/tests/codex-integration/codex-quota-auto-refresh.test.ts b/tests/codex-integration/codex-quota-auto-refresh.test.ts index 375317d9a11..a733fba73c7 100644 --- a/tests/codex-integration/codex-quota-auto-refresh.test.ts +++ b/tests/codex-integration/codex-quota-auto-refresh.test.ts @@ -226,6 +226,110 @@ describe("Codex quota window auto refresh", () => { expect(warmups).toBe(0); }); + test.each(["pool-a", "__main__"])("known deadlines need no metadata polling for %s, including after restart", async accountId => { + let cfg = config(); + cfg.codexQuotaAutoRefresh = { [accountId]: { fiveHour: true, weekly: true } }; + writeFileSync(join(testHome, "config.json"), JSON.stringify(cfg)); + let observed: StoredAccountQuota | null = quota({ + shortResetAt: RESET_SECONDS + 18_000, weeklyResetAt: RESET_SECONDS + 18_000, + updatedAt: NOW - 600_000, + }); + let probes = 0; + let warmups = 0; + const deps = { + getQuota: () => observed, + refreshQuota: async () => { probes++; }, + warmAccount: async () => { + warmups++; + observed = quota({ shortResetAt: RESET_SECONDS + 36_000, weeklyResetAt: RESET_SECONDS + 604_800 }); + }, + }; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + resetCodexQuotaAutoRefreshForTests(); + cfg = loadConfig(); + observed = null; // Durable deadlines must work without an in-memory quota snapshot. + for (let elapsed = 60_000; elapsed < 18_000_000; elapsed += 60_000) { + await runCodexQuotaAutoRefresh(cfg, NOW + elapsed, deps); + } + expect(probes).toBe(0); + expect(warmups).toBe(0); + await runCodexQuotaAutoRefresh(cfg, NOW + 18_000_000, deps); + expect(warmups).toBe(1); + expect(probes).toBe(0); + expect(loadConfig().codexQuotaAutoRefresh?.[accountId]).toMatchObject({ + lastFiveHourResetAt: NOW + 18_000_000, lastWeeklyResetAt: NOW + 18_000_000, + nextFiveHourResetAt: NOW + 36_000_000, nextWeeklyResetAt: NOW + 604_800_000, + }); + await runCodexQuotaAutoRefresh(cfg, NOW + 18_060_000, deps); + expect(probes).toBe(0); + expect(warmups).toBe(1); + }); + + test("missing-window discovery backs off to an hour and stops when passive headers supply it", async () => { + const cfg = config(); + let observed = quota({ shortResetAt: RESET_SECONDS + 86_400, weeklyResetAt: undefined, updatedAt: NOW - 600_000 }); + let probes = 0; + const deps = { + getQuota: () => observed, + refreshQuota: async () => { probes++; }, // A successful read without the missing field also backs off. + warmAccount: async () => { throw new Error("not due"); }, + }; + let elapsed = 0; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + for (const delay of [5, 10, 20, 40, 60, 60]) { + await runCodexQuotaAutoRefresh(cfg, NOW + elapsed + delay * 60_000 - 1, deps); + const before = probes; + elapsed += delay * 60_000; + await runCodexQuotaAutoRefresh(cfg, NOW + elapsed, deps); + expect(probes).toBe(before + 1); + } + expect(probes).toBe(7); + observed = { ...observed, weeklyResetAt: RESET_SECONDS + 604_800 }; + await runCodexQuotaAutoRefresh(cfg, NOW + elapsed + 60_000, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + elapsed + 3_600_000, deps); + expect(probes).toBe(7); + }); + + test("activation without next-window headers discovers the next deadline once", async () => { + const cfg = config(); + cfg.codexQuotaAutoRefresh = { "pool-a": { fiveHour: true } }; + let observed = quota(); + let probes = 0; + let warmups = 0; + const deps = { + getQuota: () => observed, + refreshQuota: async () => { + probes++; + observed = quota({ shortResetAt: RESET_SECONDS + 18_000 }); + }, + warmAccount: async () => { warmups++; }, + persistCompleted: recordMarkers, + }; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + 300_000, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + 600_000, deps); + expect(warmups).toBe(1); + expect(probes).toBe(1); + }); + + test("failed activations back off without probing known deadlines", async () => { + const cfg = config(); + let probes = 0; + let warmups = 0; + const deps = { + getQuota: () => quota({ updatedAt: NOW - 600_000 }), + refreshQuota: async () => { probes++; }, + warmAccount: async () => { warmups++; throw new Error("fixture failure"); }, + }; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + 300_000, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + 600_000, deps); + expect(warmups).toBe(2); + await runCodexQuotaAutoRefresh(cfg, NOW + 900_000, deps); + expect(warmups).toBe(3); + expect(probes).toBe(0); + }); + test("regression: inference 401 quarantines a time-valid bearer and stops retries", async () => { const cfg = config(); writePoolCredential(); From 2a49525f1d677b991e3333fec64d11d7f203e7d8 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 15:29:58 +0900 Subject: [PATCH 35/75] fix(codex): key quota activation retries to the credential generation Follow-up to #6020, from the review on that PR. Retry records carry the credential generation, so a replaced or reauthenticated credential no longer waits out its predecessor backoff, and a failure that raced a replacement is not recorded. A local native-main admission refusal retries after one minute without doubling the upstream backoff. main account unavailable stays in the growing backoff; generation keying already lets a later token start clean. --- scripts/test-layout/layout.json | 2 +- src/codex/quota-auto-refresh-state.ts | 6 +- src/codex/quota-auto-refresh.ts | 68 ++++++-- structure/providers/openai-tiers.md | 4 + ...odex-quota-auto-refresh-generation.test.ts | 148 ++++++++++++++++++ tests/fixtures/test-layout-expected.json | 1 + 6 files changed, 218 insertions(+), 11 deletions(-) create mode 100644 tests/codex-integration/codex-quota-auto-refresh-generation.test.ts diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index b2ecb246232..fe56b9a80aa 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -168,7 +168,7 @@ } }, "explicit": { - "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "project-config-warning-snapshot.test.ts": "codex-integration", + "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "project-config-warning-snapshot.test.ts": "codex-integration", "codex-quota-auto-refresh-generation.test.ts": "codex-integration", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", "responses-compaction-recovery-policy.test.ts": "responses", "plugin-loader.test.ts": "lib", "plugin-upstream-hooks.test.ts": "lib", "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", diff --git a/src/codex/quota-auto-refresh-state.ts b/src/codex/quota-auto-refresh-state.ts index 149d9b97fc4..921148bb615 100644 --- a/src/codex/quota-auto-refresh-state.ts +++ b/src/codex/quota-auto-refresh-state.ts @@ -2,7 +2,11 @@ /** Completed/due markers use epoch milliseconds; persisted legacy markers may use seconds. */ export type CodexQuotaAutoRefreshWindows = { fiveHour?: number; weekly?: number }; -export type CodexQuotaRetry = { after: number; delay: number }; +/** + * Backoff evidence for one account. `generation` names the credential the failure was observed + * under; a record from another generation says nothing about the credential in use now. + */ +export type CodexQuotaRetry = { after: number; delay: number; generation: string }; export const completedByAccount = new Map(); export const retryAfterByAccount = new Map(); diff --git a/src/codex/quota-auto-refresh.ts b/src/codex/quota-auto-refresh.ts index 53b8eb81cda..54cd196aac8 100644 --- a/src/codex/quota-auto-refresh.ts +++ b/src/codex/quota-auto-refresh.ts @@ -52,10 +52,46 @@ export interface CodexQuotaAutoRefreshRunDeps { let inFlight: Promise | null = null; +const LOCAL_BUSY_RETRY_MS = 60_000; + +/** The native main profile was claimed locally, so no upstream request was sent. */ +export class NativeMainBusyError extends Error { + constructor() { + super("native main busy"); + this.name = "NativeMainBusyError"; + } +} + +/** The credential a retry record describes: main's quota generation, or the pool record's. */ +function credentialGeneration(accountId: string): string { + return accountId === MAIN_CODEX_ACCOUNT_ID + ? `main:${getMainQuotaCredentialGeneration()}` + : `pool:${readCodexAccountRecord(accountId)?.generation ?? "none"}`; +} + +/** A retry recorded under a replaced credential is spent; drop it rather than hold the new one. */ +function liveRetry( + retries: Map, + accountId: string, + generation: string, +): CodexQuotaRetry | undefined { + const retry = retries.get(accountId); + if (retry && retry.generation !== generation) { + retries.delete(accountId); + return undefined; + } + return retry; +} + /** Back off unsuccessful discovery/activation without adding another timer. */ -function deferRetry(retries: Map, accountId: string, now: number): number { - const delay = Math.min((retries.get(accountId)?.delay ?? RETRY_MS / 2) * 2, MAX_RETRY_MS); - retries.set(accountId, { after: now + delay, delay }); +function deferRetry( + retries: Map, + accountId: string, + now: number, + generation: string, +): number { + const delay = Math.min((liveRetry(retries, accountId, generation)?.delay ?? RETRY_MS / 2) * 2, MAX_RETRY_MS); + retries.set(accountId, { after: now + delay, delay, generation }); return delay; } @@ -203,7 +239,7 @@ async function warmAccount(config: OcxConfig, accountId: string): Promise= RETRY_MS) - && (quotaRefreshAfterByAccount.get(accountId)?.after ?? 0) <= now) { - deferRetry(quotaRefreshAfterByAccount, accountId, now); + && (liveRetry(quotaRefreshAfterByAccount, accountId, credentialGeneration(accountId))?.after ?? 0) <= now) { + deferRetry(quotaRefreshAfterByAccount, accountId, now, credentialGeneration(accountId)); try { await refresh(config, accountId); } catch { /* Retry missing metadata with backoff. */ } } if (!eligible(accountId)) return; rememberWindows(config, accountId, quotaFor(accountId)); - if (hasScheduledWindows(config, accountId)) quotaRefreshAfterByAccount.delete(accountId); - if ((retryAfterByAccount.get(accountId)?.after ?? 0) > now) return; + // Backoff from a replaced credential is dropped here, so reauthenticating or rotating an + // account never waits out the failures of the credential it replaced. + const generation = credentialGeneration(accountId); + if ((liveRetry(retryAfterByAccount, accountId, generation)?.after ?? 0) > now) return; const windows = dueCodexQuotaAutoRefreshWindows(config, accountId, quotaFor(accountId), now); if (!windows) return; try { @@ -348,7 +386,19 @@ export async function runCodexQuotaAutoRefresh( persist(config, accountId, completed); rememberWindows(config, accountId, quotaFor(accountId)); } catch (error) { - const delay = deferRetry(retryAfterByAccount, accountId, now); + // A failure that raced a credential replacement describes the old credential; the next + // sweep evaluates the replacement on its own evidence. + if (credentialGeneration(accountId) !== generation) return; + if (error instanceof NativeMainBusyError) { + // Local admission refused before any upstream request: retry soon, and keep the + // upstream backoff where it was instead of doubling it. + const previous = liveRetry(retryAfterByAccount, accountId, generation); + retryAfterByAccount.set(accountId, { + after: now + LOCAL_BUSY_RETRY_MS, delay: previous?.delay ?? RETRY_MS / 2, generation, + }); + return; + } + const delay = deferRetry(retryAfterByAccount, accountId, now, generation); const account = config.codexAccounts?.find(candidate => candidate.id === accountId); const label = account ? codexAccountLogLabel(account) : "main"; console.warn(`[codex-quota-auto-refresh] ${label}: ${codexWarmupFailureReason(error)}; ${ diff --git a/structure/providers/openai-tiers.md b/structure/providers/openai-tiers.md index 00187126f80..ad11efe96ec 100644 --- a/structure/providers/openai-tiers.md +++ b/structure/providers/openai-tiers.md @@ -215,6 +215,10 @@ Inference 401s quarantine the rejected credential; failures log an opaque label Paused or reauthentication-required accounts are skipped, simultaneous 5-hour/weekly resets share one warmup, transient activation failures back off from five minutes to an hour, and account deletion removes settings and retry/completion state. +Retry records name the credential generation they were observed under (main quota generation, pool +record generation). A record from a replaced or reauthenticated credential is dropped when read, and a +failure that raced a replacement is not recorded. A local `NativeMainBusyError` admission refusal sends +nothing upstream, so it retries after one minute and keeps the upstream backoff unchanged. Main-account hard-lock also gates these billable warmups. A policy/identity skip changes neither completion markers nor retry delay; quota reads remain available. Main refresh completes before shared credential ownership, then prepared credentials and restrictions are rechecked. Lifecycle diff --git a/tests/codex-integration/codex-quota-auto-refresh-generation.test.ts b/tests/codex-integration/codex-quota-auto-refresh-generation.test.ts new file mode 100644 index 00000000000..37a3ea6c96f --- /dev/null +++ b/tests/codex-integration/codex-quota-auto-refresh-generation.test.ts @@ -0,0 +1,148 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { existsSync, mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { + NativeMainBusyError, + resetCodexQuotaAutoRefreshForTests, + runCodexQuotaAutoRefresh, + type CodexQuotaAutoRefreshWindows, +} from "../../src/codex/quota-auto-refresh"; +import { clearAccountQuota, type StoredAccountQuota } from "../../src/codex/quota"; +import { readCodexAccountRecord, saveCodexAccountCredential } from "../../src/codex/account-store"; +import type { OcxConfig } from "../../src/types"; + +/** + * Retry evidence belongs to the credential it was observed under (#6020 review). A replaced or + * reauthenticated credential must not wait out its predecessor's backoff, and a local admission + * refusal that sent nothing upstream must not grow the upstream backoff. + */ +const NOW = 1_800_000_000_000; +const RESET_SECONDS = NOW / 1000; +let testHome = ""; +let previousHome: string | undefined; + +function writePoolCredential(accessToken: string): void { + saveCodexAccountCredential("pool-a", { + accessToken, refreshToken: "generation-refresh-fixture", + expiresAt: NOW + 86_400_000, chatgptAccountId: "generation-workspace-fixture", + }); +} + +function config(): OcxConfig { + return { + defaultProvider: "openai", + providers: { openai: { + adapter: "openai-responses", baseUrl: "https://chatgpt.com/backend-api/codex", + authMode: "forward", codexAccountMode: "pool", + } }, + codexAccounts: [{ id: "pool-a", email: "p***a@example.test", plan: "team", isMain: false }], + codexQuotaAutoRefresh: { "pool-a": { fiveHour: true, weekly: true } }, + }; +} + +function quota(): StoredAccountQuota { + return { shortWindowSeconds: 5 * 60 * 60, shortResetAt: RESET_SECONDS, weeklyResetAt: RESET_SECONDS, updatedAt: NOW }; +} + +function recordMarkers(cfg: OcxConfig, accountId: string, completed: CodexQuotaAutoRefreshWindows): boolean { + cfg.codexQuotaAutoRefresh = { ...cfg.codexQuotaAutoRefresh, [accountId]: { + ...cfg.codexQuotaAutoRefresh?.[accountId], + ...(completed.fiveHour !== undefined ? { lastFiveHourResetAt: completed.fiveHour } : {}), + ...(completed.weekly !== undefined ? { lastWeeklyResetAt: completed.weekly } : {}), + } }; + return true; +} + +beforeEach(() => { + previousHome = process.env.OPENCODEX_HOME; + testHome = mkdtempSync(join(tmpdir(), "ocx-quota-generation-")); + process.env.OPENCODEX_HOME = testHome; + clearAccountQuota(); + resetCodexQuotaAutoRefreshForTests(); + writePoolCredential("generation-one"); +}); + +afterEach(() => { + clearAccountQuota(); + resetCodexQuotaAutoRefreshForTests(); + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + if (testHome && existsSync(testHome)) rmSync(testHome, { recursive: true, force: true }); +}); + +describe("quota auto-refresh retry evidence follows the credential", () => { + test("a replaced credential does not wait out its predecessor's backoff", async () => { + const cfg = config(); + let attempts = 0; + const deps = { + getQuota: () => quota(), + warmAccount: async () => { attempts++; if (attempts === 1) throw new Error("fixture failure"); }, + persistCompleted: recordMarkers, + }; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + expect(attempts).toBe(1); + // Same credential: the five-minute backoff holds. + await runCodexQuotaAutoRefresh(cfg, NOW + 60_000, deps); + expect(attempts).toBe(1); + const before = readCodexAccountRecord("pool-a")?.generation; + writePoolCredential("generation-two"); + expect(readCodexAccountRecord("pool-a")?.generation).not.toBe(before); + // Replacement: the old backoff is spent and the new credential is tried at once. + await runCodexQuotaAutoRefresh(cfg, NOW + 120_000, deps); + expect(attempts).toBe(2); + expect(cfg.codexQuotaAutoRefresh?.["pool-a"]?.lastFiveHourResetAt).toBe(NOW); + }); + + test("a failure that raced a replacement is not recorded against the new credential", async () => { + const cfg = config(); + let attempts = 0; + const deps = { + getQuota: () => quota(), + warmAccount: async () => { + attempts++; + if (attempts === 1) { + writePoolCredential("rotated-during-warmup"); + throw new Error("fixture failure from the old credential"); + } + }, + persistCompleted: recordMarkers, + }; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + expect(cfg.codexQuotaAutoRefresh?.["pool-a"]?.lastFiveHourResetAt).toBeUndefined(); + await runCodexQuotaAutoRefresh(cfg, NOW + 60_000, deps); + expect(attempts).toBe(2); + expect(cfg.codexQuotaAutoRefresh?.["pool-a"]?.lastFiveHourResetAt).toBe(NOW); + }); + + test("a local busy refusal retries every minute without growing the upstream backoff", async () => { + const cfg = config(); + let attempts = 0; + let outcome: "busy" | "fail" | "ok" = "busy"; + const deps = { + getQuota: () => quota(), + warmAccount: async () => { + attempts++; + if (outcome === "busy") throw new NativeMainBusyError(); + if (outcome === "fail") throw new Error("fixture upstream failure"); + }, + persistCompleted: recordMarkers, + }; + await runCodexQuotaAutoRefresh(cfg, NOW, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + 60_000 - 1, deps); + expect(attempts).toBe(1); + await runCodexQuotaAutoRefresh(cfg, NOW + 60_000, deps); + await runCodexQuotaAutoRefresh(cfg, NOW + 120_000, deps); + expect(attempts).toBe(3); + // The first real upstream failure starts from the base five minutes, not a doubled delay. + outcome = "fail"; + await runCodexQuotaAutoRefresh(cfg, NOW + 180_000, deps); + expect(attempts).toBe(4); + outcome = "ok"; + await runCodexQuotaAutoRefresh(cfg, NOW + 180_000 + 5 * 60_000 - 1, deps); + expect(attempts).toBe(4); + await runCodexQuotaAutoRefresh(cfg, NOW + 180_000 + 5 * 60_000, deps); + expect(attempts).toBe(5); + expect(cfg.codexQuotaAutoRefresh?.["pool-a"]?.lastFiveHourResetAt).toBe(NOW); + }); +}); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 97b9abfedc3..04cfd1ace90 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1,6 +1,7 @@ { "pnpm-command-isolation.test.ts": "update", "project-config-warning-snapshot.test.ts": "codex-integration", + "codex-quota-auto-refresh-generation.test.ts": "codex-integration", "provider-antigravity-quota-retry.test.ts": "providers", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", From 9fdc0c458241f215f13bd0124bcaca1711693a4f Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 15:30:39 +0900 Subject: [PATCH 36/75] fix(combos): hold a spent token-plan window for ten minutes Fixes #5494. A 429 saying "Your token-plan 1-week quota has been exhausted" took the 60-second default cooldown, so the combo offered the dead target again every minute; with one transient failure on the other target, every request in between got 503 No available targets. The phrase now counts as account-window exhaustion, which holds the target for the existing ten-minute cap. The match is anchored to the token-plan wording so per-minute quota messages keep the short cooldown. --- src/combos/failover.ts | 5 ++++- structure/runtime.md | 2 +- .../combo-codex-exhaustion-cooldown.test.ts | 17 +++++++++++++++++ 3 files changed, 22 insertions(+), 2 deletions(-) diff --git a/src/combos/failover.ts b/src/combos/failover.ts index dac6cdeeefd..da657891bd8 100644 --- a/src/combos/failover.ts +++ b/src/combos/failover.ts @@ -326,7 +326,10 @@ function normalizedFailureCode(code?: string | null): string { // are quota-limit codes whose window length this gateway has no evidence for, and guessing long on // them would hold a target that may clear sooner. const ACCOUNT_EXHAUSTION_CODES = new Set(["usage_limit_exceeded", "usage_limit_reached", "1308"]); -const ACCOUNT_EXHAUSTION_TEXT = /usage limit (?:has been )?reached/; +// Token-plan windows (Alibaba's DeepSeek/Qwen plans) report "Your token-plan 1-week quota has been +// exhausted" (#5494). The match is anchored to that phrasing: a looser "quota ... exhausted" would +// also catch per-minute limits, and this arm outranks the transient rate-limit duration. +const ACCOUNT_EXHAUSTION_TEXT = /usage limit (?:has been )?reached|token-plan\s+\S+\s+quota has been exhausted/; function isAccountWindowExhausted(message: string, code?: string | null): boolean { return ACCOUNT_EXHAUSTION_CODES.has(normalizedFailureCode(code)) diff --git a/structure/runtime.md b/structure/runtime.md index 98ddd7ccd46..8da1d766045 100644 --- a/structure/runtime.md +++ b/structure/runtime.md @@ -504,7 +504,7 @@ This is also why the classifier cannot duplicate visible output. Native byte str Regression coverage: `tests/responses/responses-forward-prompt-envelope.test.ts`, `tests/routing/router-combo-failover-classification.test.ts`, `tests/routing/routing-policy-fallback.test.ts`, `tests/helpers/combo-context-overflow-cases.ts`, and `tests/server/server-combo-failover-e2e.test.ts`. -`src/combos/failover.ts` uses a 10-minute fallback for a spent account usage window (codes `usage_limit_exceeded`, `usage_limit_reached`, `1308`, or `usage limit reached` / `usage limit has been reached` prose, including HTTP 502) and for provider-scoped credential or billing failure codes such as `invalid_api_key` and `insufficient_quota`. This duration does not change failure classification or cooldown scope; upstream retry/reset signals and configured durations retain precedence. It caps explicit upstream `Retry-After` target cooldowns at 24 hours while reset-derived, configured, and fallback cooldowns remain capped at 10 minutes. +`src/combos/failover.ts` uses a 10-minute fallback for a spent account usage window (codes `usage_limit_exceeded`, `usage_limit_reached`, `1308`, or `usage limit reached` / `usage limit has been reached` / `token-plan quota has been exhausted` prose, including HTTP 502) and for provider-scoped credential or billing failure codes such as `invalid_api_key` and `insufficient_quota`. This duration does not change failure classification or cooldown scope; upstream retry/reset signals and configured durations retain precedence. It caps explicit upstream `Retry-After` target cooldowns at 24 hours while reset-derived, configured, and fallback cooldowns remain capped at 10 minutes. ## Combo default effort precedence diff --git a/tests/codex-integration/combo-codex-exhaustion-cooldown.test.ts b/tests/codex-integration/combo-codex-exhaustion-cooldown.test.ts index ddb882ed9ab..0aa831344c8 100644 --- a/tests/codex-integration/combo-codex-exhaustion-cooldown.test.ts +++ b/tests/codex-integration/combo-codex-exhaustion-cooldown.test.ts @@ -53,6 +53,23 @@ describe("a depleted Codex plan window", () => { expect(isComboTargetInCooldown(combo, target, now + 60_000)).toBe(false); }); + // #5494: a token-plan 429 is a spent plan window. At 60s the combo re-offered it every minute and, + // with one transient failure on the other target, answered every request with 503. + test("a spent token-plan window takes the ten-minute hold", () => { + const message = "Provider error 429: Your token-plan 1-week quota has been exhausted. " + + "The quota will reset at 10-04 12:00:00 UTC."; + coolComboTarget(combo, target, { now, status: 429, code: "rate_limit_exceeded", message }); + expect(isComboTargetInCooldown(combo, target, now + 10 * 60_000 - 1)).toBe(true); + expect(isComboTargetInCooldown(combo, target, now + 10 * 60_000)).toBe(false); + }); + + test("a per-minute quota message keeps the short cooldown", () => { + coolComboTarget(combo, target, { + now, status: 429, code: "rate_limit_exceeded", message: "rate limit exceeded, quota exhausted for this minute", + }); + expect(isComboTargetInCooldown(combo, target, now + 60_000)).toBe(false); + }); + test("a bare 1308 code with no prose still takes the exhaustion hold", () => { coolComboTarget(combo, target, { now, status: 429, code: "1308", message: "" }); expect(isComboTargetInCooldown(combo, target, now + 10 * 60_000 - 1)).toBe(true); From cafe6202ad4c5f77fa0c6d9eae49f444521c0944 Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 15:31:20 +0900 Subject: [PATCH 37/75] fix(cli): disambiguate auto account selection (#6050) Carried from #6050 into merge train round 3. Clearing the selection now succeeds while main is paused, so the dev test that pinned the old 409 is removed; codex-account-clear-paused.test.ts covers the new contract, including that an explicit paused-main selection still gets 409. Co-authored-by: Epinephrine --- .../fr/reference/cli/providers-accounts.md | 11 +++- .../ja/reference/cli/providers-accounts.md | 11 +++- .../ko/reference/cli/providers-accounts.md | 11 +++- .../docs/reference/cli/providers-accounts.md | 11 +++- .../ru/reference/cli/providers-accounts.md | 11 +++- .../tr/reference/cli/providers-accounts.md | 11 +++- .../zh-cn/reference/cli/providers-accounts.md | 11 +++- .../zh-tw/reference/cli/providers-accounts.md | 11 +++- scripts/test-layout/layout.json | 2 +- src/cli/account-target.ts | 19 +++--- src/cli/account.ts | 47 ++++++++++++++- src/codex/auth-api/routes.ts | 2 +- structure/providers/openai-accounts.md | 7 ++- tests/cli/cli-account-alias-target.test.ts | 58 +++++++++++++++---- .../codex-account-clear-paused.test.ts | 55 ++++++++++++++++++ .../codex-integration/codex-auth-api.test.ts | 14 ----- tests/fixtures/test-layout-expected.json | 1 + 17 files changed, 232 insertions(+), 61 deletions(-) create mode 100644 tests/codex-integration/codex-account-clear-paused.test.ts diff --git a/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md b/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md index 9282cf080f0..3fb8774a8ce 100644 --- a/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md @@ -105,12 +105,13 @@ Répertoriez et changez de compte de fournisseur et de pools de clés API via le la surface est : ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -192,7 +193,7 @@ cet état et quitte toujours 0. `--json` renvoie : ### `ocx account use [--json]` -`auto` efface la sélection manuelle pour que le pool place à nouveau le travail selon sa propre stratégie. Un compte Codex peut être désigné par l'alias défini avec `ocx account alias` au lieu de son id ; cela vaut aussi pour `priority`, `pause`, `resume`, `clear-cooldown`, `remove` et `alias`. Pour les comptes Codex, `auto`, `main` et `__main__` sont réservés sans distinction de casse et ne peuvent pas être attribués comme alias. Les noms affichés des comptes OAuth et des clés API conservent leurs règles existantes. +`auto` efface la sélection manuelle pour que le pool place à nouveau le travail selon sa propre stratégie — sauf si un compte Codex porte littéralement l'id `auto`, qui l'emporte par correspondance exacte d'id ; `ocx account clear ` rétablit toujours la sélection automatique. Un compte Codex peut être désigné par l'alias défini avec `ocx account alias` au lieu de son id ; cela vaut aussi pour `priority`, `pause`, `resume`, `clear-cooldown`, `remove` et `alias`. Pour les comptes Codex, `auto`, `main` et `__main__` sont réservés sans distinction de casse et ne peuvent pas être attribués comme alias. Les noms affichés des comptes OAuth et des clés API conservent leurs règles existantes. Sélectionne un compte Codex, un compte OAuth ou une clé API existant. Pour `openai`, `main` sélectionne la connexion Codex App. Une sélection en mode Codex Pool efface l'affinité locale du processus et s'applique à la requête suivante, @@ -213,6 +214,10 @@ faire basculer la requête vers un autre compte de pool admissible. Ces transiti { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +Efface la sélection manuelle du compte Codex sans résoudre d'id de compte, donc fonctionne même lorsqu'un compte s'appelle littéralement `auto`. Pools Codex uniquement ; les autres types de fournisseur n'ont pas de sélection automatique à rétablir. + ### `ocx account refresh [--json]` Pour le groupe de comptes Codex, utilisez `ocx account refresh openai [--json]`. Cette commande force l'actualisation des quotas de compte et diff --git a/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md index 2eb215a1188..479fac0f0da 100644 --- a/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md @@ -79,12 +79,13 @@ ocx login anthropic 実行中のプロキシを介してプロバイダー アカウントと API キー プールを一覧表示し、切り替えます。出荷されたヘルプ画面は次のとおりです。 ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -144,7 +145,7 @@ Codex pool selection applies to the next request after clearing existing affinit ### `ocx account use [--json]` -`auto` は手動の選択を解除し、プールが自身の戦略で再び配置するようにします。Codex アカウントは id の代わりに `ocx account alias` で付けたエイリアスでも指定でき、`priority`、`pause`、`resume`、`clear-cooldown`、`remove`、`alias` でも同様です。Codex アカウントでは `auto`、`main`、`__main__` は大文字・小文字を区別せず予約語として扱われるため、エイリアスとして設定できません。OAuth アカウントと API キーの表示名には従来のルールが適用されます。 +`auto` は手動の選択を解除し、プールが自身の戦略で再び配置するようにします — ただし id が `auto` の Codex アカウントが存在する場合は完全一致の id が優先され、`ocx account clear ` が常に自動選択を復元します。Codex アカウントは id の代わりに `ocx account alias` で付けたエイリアスでも指定でき、`priority`、`pause`、`resume`、`clear-cooldown`、`remove`、`alias` でも同様です。Codex アカウントでは `auto`、`main`、`__main__` は大文字・小文字を区別せず予約語として扱われるため、エイリアスとして設定できません。OAuth アカウントと API キーの表示名には従来のルールが適用されます。 既存の Codex アカウント、OAuth アカウント、または API key を選びます。`openai` で `main` は Codex App ログインを 選択します。Codex Pool の選択は process-local affinity を消去し、既存の表示タスクを含む次のリクエストから適用されます。プロキシ再起動や affinity eviction 後もタスクは未紐付けになり得ますが、処理中のリクエストは取得済みアカウントを維持します。この選択は Pool routing のみを制御し、Direct mode は caller-owned/native main credential を使い続けます。使用量ベースのプロアクティブ切り替え、401/403 再認証、429/retry-after cooldown、除外、出力前 429/402 の障害回復により、後で別の適格 Pool アカウントが選ばれる場合があります。これらの回復経路は使用量ベース切り替えが off でも有効です。アカウント変更後も OpenCodex は会話コンテキストを再生しますが、provider prompt cache は再ウォームアップが必要な場合があります。 @@ -158,6 +159,10 @@ Codex pool selection applies to the next request after clearing existing affinit { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +アカウント id を解決せずに Codex アカウントの手動選択を解除するため、`auto` という id のアカウントが存在しても機能します。Codex プール専用です。他のプロバイダー種別には復元する自動選択がありません。 + ### `ocx account refresh [--json]` Codex プールの場合は、`ocx account refresh openai [--json]` を使用します。アカウント クォータを強制的に更新し、利用可能な週次/月次のパーセンテージとリセット時間を出力します。不足しているクォータ データは、0% ではなく不明として報告されます。その JSON エンベロープは `{ accounts: AccountRow[] }` で、Codex の各行に `quota` があります。 diff --git a/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md index 5f213530fab..3a395cca85d 100644 --- a/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md @@ -164,12 +164,13 @@ Luna 메타데이터임을 표시해 사용합니다. 목록에 보인다는 사 실행 중인 프록시를 통해 제공자 계정과 API 키 풀을 나열하고 전환합니다. 제공되는 도움말 표면은 다음과 같습니다: ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -229,7 +230,7 @@ Codex pool selection applies to the next request after clearing existing affinit ### `ocx account use [--json]` -`auto`는 수동 선택을 지워 풀이 다시 자체 전략으로 작업을 배치하게 합니다. Codex 계정은 id 대신 `ocx account alias`로 지정한 별칭으로도 가리킬 수 있으며, `priority`, `pause`, `resume`, `clear-cooldown`, `remove`, `alias`에서도 마찬가지입니다. Codex 계정에서 `auto`, `main`, `__main__`은 대소문자 구분 없이 예약어이므로 별칭으로 지정할 수 없습니다. OAuth 계정과 API 키의 표시 이름에는 기존 규칙이 그대로 적용됩니다. +`auto`는 수동 선택을 지워 풀이 다시 자체 전략으로 작업을 배치하게 합니다 — 단 id가 `auto`인 Codex 계정이 있으면 정확한 id 일치가 우선되며 `ocx account clear `는 항상 자동 선택을 복원합니다. Codex 계정은 id 대신 `ocx account alias`로 지정한 별칭으로도 가리킬 수 있으며, `priority`, `pause`, `resume`, `clear-cooldown`, `remove`, `alias`에서도 마찬가지입니다. Codex 계정에서 `auto`, `main`, `__main__`은 대소문자 구분 없이 예약어이므로 별칭으로 지정할 수 없습니다. OAuth 계정과 API 키의 표시 이름에는 기존 규칙이 그대로 적용됩니다. 기존 Codex 계정, OAuth 계정 또는 API key를 선택합니다. `openai`에서 `main`은 Codex App 로그인을 선택합니다. Codex Pool 선택은 프로세스 로컬 affinity를 지우고 기존에 보이던 작업을 포함한 다음 요청부터 적용됩니다. 프록시 재시작이나 affinity eviction 뒤에도 작업이 바인딩 없는 상태가 될 수 있지만, 진행 중인 요청은 이미 확보한 계정을 유지합니다. 이 선택은 Pool 라우팅만 제어하며 Direct mode는 호출자 소유/native main credential을 계속 사용합니다. 사용량 기반 선제 전환, 401/403 재인증, 429/retry-after cooldown, 제외, 출력 전 429/402 실패 복구는 나중에 다른 적격 Pool 계정을 선택할 수 있습니다. 이러한 복구 경로는 사용량 기반 전환이 꺼져 있어도 동작합니다. 계정이 바뀌어도 OpenCodex는 대화 문맥을 재생하지만 프로바이더 측 prompt cache는 다시 예열해야 할 수 있습니다. @@ -243,6 +244,10 @@ Codex pool selection applies to the next request after clearing existing affinit { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +계정 id를 해석하지 않고 Codex 계정의 수동 선택을 지우므로 `auto`라는 id의 계정이 있어도 동작합니다. Codex 풀 전용이며 다른 공급자 유형에는 복원할 자동 선택이 없습니다. + ### `ocx account refresh [--json]` Codex 풀에는 `ocx account refresh openai [--json]`를 사용합니다. 계정 할당량을 강제로 새로 고치고 사용 가능 주간/월간 비율과 재설정 시간을 출력합니다. 할당량 데이터가 없으면 0%가 아니라 알 수 없음으로 보고합니다. JSON 봉투는 `{ accounts: AccountRow[] }`이며, Codex 행마다 `quota`가 붙습니다. diff --git a/docs-site/src/content/docs/reference/cli/providers-accounts.md b/docs-site/src/content/docs/reference/cli/providers-accounts.md index 9805136a375..93abe731d2e 100644 --- a/docs-site/src/content/docs/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/reference/cli/providers-accounts.md @@ -226,12 +226,13 @@ List and switch provider accounts and API-key pools through the running proxy. T surface is: ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -420,7 +421,7 @@ that state and still exits 0. `--json` returns: ### `ocx account use [--json]` -`auto` clears the manual selection so the pool places work by its own strategy again. Any Codex account can be named by the alias set with `ocx account alias` instead of its id; that holds for `priority`, `pause`, `resume`, `clear-cooldown`, `remove` and `alias` too. For Codex accounts, `auto`, `main` and `__main__` are reserved regardless of case and cannot be assigned as aliases. OAuth and API-key display names keep their existing rules. +`auto` clears the manual selection so the pool places work by its own strategy again — unless a Codex account literally carries the id `auto`, which wins by exact-id precedence; `ocx account clear ` always restores automatic selection. Any Codex account can be named by the alias set with `ocx account alias` instead of its id; that holds for `priority`, `pause`, `resume`, `clear-cooldown`, `remove` and `alias` too. For Codex accounts, `auto`, `main` and `__main__` are reserved regardless of case and cannot be assigned as aliases. OAuth and API-key display names keep their existing rules. Selects an existing Codex account, OAuth account, or API key. For `openai`, `main` selects the Codex App login. A Codex Pool selection clears process-local affinity and applies to the next request, @@ -441,6 +442,10 @@ rotate the request to another eligible Pool account. These failure transitions r { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +Clear the manual Codex account selection without resolving an account id, so it works even when an account is literally named `auto`. Codex pools only; other provider types have no automatic selection to restore. + ### `ocx account refresh [--json]` For the Codex pool, use `ocx account refresh openai [--json]`. It force-refreshes account quotas and diff --git a/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md index 8bf82147136..7339d3f53bb 100644 --- a/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md @@ -94,12 +94,13 @@ ocx login anthropic Поставляемая help-surface выглядит так: ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -175,7 +176,7 @@ credential'а, это состояние тоже печатается, но к ### `ocx account use [--json]` -`auto` снимает ручной выбор, и пул снова распределяет работу по своей стратегии. Аккаунт Codex можно указать по псевдониму, заданному через `ocx account alias`, вместо id; это относится и к `priority`, `pause`, `resume`, `clear-cooldown`, `remove` и `alias`. Для аккаунтов Codex значения `auto`, `main` и `__main__` зарезервированы независимо от регистра и не могут назначаться как псевдонимы. Для отображаемых имён аккаунтов OAuth и API-ключей действуют прежние правила. +`auto` снимает ручной выбор, и пул снова распределяет работу по своей стратегии — если только аккаунт Codex буквально не имеет id `auto`: точное совпадение id выигрывает, а `ocx account clear ` всегда восстанавливает автоматический выбор. Аккаунт Codex можно указать по псевдониму, заданному через `ocx account alias`, вместо id; это относится и к `priority`, `pause`, `resume`, `clear-cooldown`, `remove` и `alias`. Для аккаунтов Codex значения `auto`, `main` и `__main__` зарезервированы независимо от регистра и не могут назначаться как псевдонимы. Для отображаемых имён аккаунтов OAuth и API-ключей действуют прежние правила. Выбирает существующий аккаунт Codex, OAuth-аккаунт или API-ключ. Для `openai` значение `main` выбирает вход Codex App. Выбор Codex Pool очищает process-local affinity и применяется к следующему запросу, включая запрос существующей видимой задачи; после перезапуска прокси или affinity eviction задача также может стать непривязанной, а выполняющиеся запросы сохраняют захваченный аккаунт. Это управляет только Pool routing; Direct mode продолжает использовать caller-owned/native main credential. Проактивное переключение по использованию, повторная аутентификация 401/403, cooldown 429/retry-after, исключение и восстановление после отказа 429/402 до вывода могут позже выбрать другой подходящий Pool-аккаунт. Эти пути восстановления остаются активными, когда переключение по использованию выключено. После смены аккаунта OpenCodex воспроизводит контекст разговора, но prompt cache провайдера может потребовать прогрева. Неизвестные провайдеры @@ -189,6 +190,10 @@ credential'а, это состояние тоже печатается, но к { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +Снимает ручной выбор аккаунта Codex без разрешения id, поэтому работает, даже когда аккаунт буквально называется `auto`. Только для пулов Codex; у других типов провайдеров нет автоматического выбора для восстановления. + ### `ocx account refresh [--json]` Для пула Codex используйте `ocx account refresh openai [--json]`. Команда принудительно diff --git a/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md b/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md index a437c1042bb..1e5e797f963 100644 --- a/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md @@ -113,12 +113,13 @@ Bir sağlayıcı için saklanan OAuth kimlik bilgisini kaldırın. listeleyin ve değiştirin. Sağlanan yardım arayüzü şöyledir: ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -207,7 +208,7 @@ yine de 0 ile çıkar. `--json` şunu döndürür: ### `ocx account use [--json]` -`auto` elle yapılan seçimi temizler; havuz işi yeniden kendi stratejisiyle yerleştirir. Bir Codex hesabı, id yerine `ocx account alias` ile verilen takma adla da belirtilebilir; bu `priority`, `pause`, `resume`, `clear-cooldown`, `remove` ve `alias` için de geçerlidir. Codex hesaplarında `auto`, `main` ve `__main__` büyük/küçük harf fark etmeksizin ayrılmış sözcüklerdir ve takma ad olarak atanamaz. OAuth hesaplarının ve API anahtarlarının görünen adları için mevcut kurallar geçerlidir. +`auto` elle yapılan seçimi temizler; havuz işi yeniden kendi stratejisiyle yerleştirir — ancak id'si `auto` olan bir Codex hesabı varsa tam id eşleşmesi kazanır ve `ocx account clear ` her zaman otomatik seçimi geri yükler. Bir Codex hesabı, id yerine `ocx account alias` ile verilen takma adla da belirtilebilir; bu `priority`, `pause`, `resume`, `clear-cooldown`, `remove` ve `alias` için de geçerlidir. Codex hesaplarında `auto`, `main` ve `__main__` büyük/küçük harf fark etmeksizin ayrılmış sözcüklerdir ve takma ad olarak atanamaz. OAuth hesaplarının ve API anahtarlarının görünen adları için mevcut kurallar geçerlidir. Mevcut bir Codex hesabını, OAuth hesabını veya API anahtarını seçer. `openai` için `main` Codex App girişini seçer. Bir Codex Havuzu seçimi süreç içi yerel @@ -234,6 +235,10 @@ ayar yalnızca kullanıma dayalı proaktif geçişi devre dışı bırakır. { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +Bir hesap id'si çözümlemeden Codex hesabının elle seçimini temizler; `auto` adında bir hesap olsa bile çalışır. Yalnızca Codex havuzları içindir; diğer sağlayıcı türlerinde geri yüklenecek otomatik seçim yoktur. + ### `ocx account refresh [--json]` Codex havuzu için `ocx account refresh openai [--json]` kullanın. Hesap diff --git a/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md b/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md index de6092ef109..2a102ecec5c 100644 --- a/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md @@ -84,12 +84,13 @@ ocx login anthropic 通过正在运行的代理列出并切换提供方账号和 API 密钥池。随附的帮助输出如下: ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex account pool, OAuth accounts and API keys (identifiers shown masked as the API returns them). history openai [--limit <1-200>] Recent routing decisions for one Codex pool account. current Show the active account or key. -use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection. +use Switch the active credential; 'main' selects the Codex App login, 'auto' clears the selection unless an account carries that id. +clear Clear the manual Codex account selection unconditionally. refresh Force-refresh Codex or provider quota reports. auto-switch Control the Codex pool threshold. alias Set or clear an account's display name; '-' clears it. @@ -159,7 +160,7 @@ OAuth 账号会显示为 `Account N`,而 plan/label 列会在 plan、屏蔽后 ### `ocx account use [--json]` -`auto` 会清除手动选择,让 Pool 重新按自身策略分配工作。Codex 账号可以用 `ocx account alias` 设置的别名代替 id 来指定;`priority`、`pause`、`resume`、`clear-cooldown`、`remove` 和 `alias` 同样如此。对于 Codex 账号,`auto`、`main` 和 `__main__` 为保留字(不区分大小写),不能设为别名。OAuth 账号和 API 密钥的显示名称仍遵循原有规则。 +`auto` 会清除手动选择,让 Pool 重新按自身策略分配工作 — 但如果某个 Codex 账号的 id 恰为 `auto`,则精确 id 匹配优先;`ocx account clear ` 始终恢复自动选择。Codex 账号可以用 `ocx account alias` 设置的别名代替 id 来指定;`priority`、`pause`、`resume`、`clear-cooldown`、`remove` 和 `alias` 同样如此。对于 Codex 账号,`auto`、`main` 和 `__main__` 为保留字(不区分大小写),不能设为别名。OAuth 账号和 API 密钥的显示名称仍遵循原有规则。 选择已有的 Codex 账号、OAuth 账号或 API key。对 `openai` 而言,`main` 选择 Codex App 登录。 Codex Pool 选择会清除进程本地 affinity,并从下一次请求开始生效,包括已有可见任务的请求;代理重启或 affinity eviction 后,任务也可能变为未绑定,但进行中的请求保留已捕获账号。此选择只控制 Pool routing;Direct mode 继续使用 caller-owned/native main credential。基于用量的主动切换、401/403 重新认证、429/retry-after cooldown、排除,以及输出前 429/402 故障恢复之后仍可能选择其他合格 Pool 账号。这些恢复路径在关闭基于用量的切换时仍然有效。账号变化后 OpenCodex 会重放对话上下文,但 provider prompt cache 可能需要重新预热。未知 provider 或 id 返回退出码 1。`--json` 返回: @@ -172,6 +173,10 @@ Codex Pool 选择会清除进程本地 affinity,并从下一次请求开始生 { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +在不解析账号 id 的情况下清除 Codex 账号的手动选择,因此即使存在名为 `auto` 的账号也有效。仅适用于 Codex Pool;其他提供商类型没有可恢复的自动选择。 + ### `ocx account refresh [--json]` 对于 Codex 池,请使用 `ocx account refresh openai [--json]`。它会强制刷新账号配额, diff --git a/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md b/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md index 59d373ee179..432f525737f 100644 --- a/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md @@ -62,12 +62,13 @@ ocx login anthropic 透過執行中的代理列出並切換供應商帳號與 API-key 池。隨附的說明介面如下: ```text -Usage: ocx account ... +Usage: ocx account ... list [provider] Codex 帳號池、OAuth 帳號與 API 金鑰(識別碼依 API 回傳遮罩顯示)。 history openai [--limit <1-200>] 單一 Codex 帳號池帳號的近期路由決策。 current 顯示現用帳號或金鑰。 -use 切換現用憑證;'main' 選擇 Codex App 登入,'auto' 清除選擇。 +use 切換現用憑證;'main' 選擇 Codex App 登入,'auto' 清除選擇,除非有帳號的 id 恰為此值。 +clear 無條件清除 Codex 帳號的手動選擇。 refresh 強制重新整理 Codex 或供應商配額報告。 auto-switch 控制 Codex 池閾值。 alias 設定或清除帳號顯示名稱;'-' 表示清除。 @@ -126,7 +127,7 @@ Codex 池選擇套用於清除既有親和性後的下一個請求;進行中 ### `ocx account use [--json]` -`auto` 會清除手動選擇,讓池重新依自身策略分配工作。Codex 帳號可以用 `ocx account alias` 設定的別名代替 id 來指定;`priority`、`pause`、`resume`、`clear-cooldown`、`remove` 與 `alias` 亦然。對於 Codex 帳號,`auto`、`main` 和 `__main__` 為保留字(不區分大小寫),不能設為別名。OAuth 帳號與 API 金鑰的顯示名稱仍遵循原有規則。 +`auto` 會清除手動選擇,讓池重新依自身策略分配工作 — 但若 Codex 帳號的 id 恰為 `auto`,則精確 id 比對優先;`ocx account clear ` 一律還原自動選擇。Codex 帳號可以用 `ocx account alias` 設定的別名代替 id 來指定;`priority`、`pause`、`resume`、`clear-cooldown`、`remove` 與 `alias` 亦然。對於 Codex 帳號,`auto`、`main` 和 `__main__` 為保留字(不區分大小寫),不能設為別名。OAuth 帳號與 API 金鑰的顯示名稱仍遵循原有規則。 選擇既有的 Codex 帳號、OAuth 帳號或 API 金鑰。對於 `openai`,`main` 選擇 Codex App 登入。Codex 池選擇清除行程本地親和性並套用於下一個請求,包含來自既有可見任務的請求;代理重啟或親和性驅逐也可能使任務未綁定,而進行中的請求保留其擷取的帳號。這僅控制池路由;Direct 模式繼續使用呼叫者擁有/原生的 main 憑證。基於用量的主動切換、401/403 重新認證、429/retry-after 冷卻、排除,以及 pre-output 429/402 失敗復原稍後可能選擇另一個合格的池帳號。當基於用量的切換關閉時,這些復原路徑仍然活躍。OpenCodex 在帳號變更後重播對話,但供應商端的 prompt cache 可能是冷的。未知的供應商或 id 離開 1。 在 **401/403** 時,App 登入清除該帳號的行程本地親和性並要求重新認證。 @@ -137,6 +138,10 @@ Codex 池選擇套用於清除既有親和性後的下一個請求;進行中 { ok: true, provider, type, activeId } ``` +### `ocx account clear [--json]` + +不解析帳號 id 即清除 Codex 帳號的手動選擇,即使存在名為 `auto` 的帳號仍有效。僅適用於 Codex 池;其他提供者類型沒有可還原的自動選擇。 + ### `ocx account refresh [--json]` 對於 Codex 池,請使用 `ocx account refresh openai [--json]`。它強制重新整理帳號配額並印出可用的週/月百分比與重置時間;缺失的配額資料被回報為未知,而非 0%。其 JSON 封裝為 `{ accounts: AccountRow[] }`,每個 Codex 列上有 `quota`。 diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index fe56b9a80aa..53bcb676449 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -168,7 +168,7 @@ } }, "explicit": { - "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "project-config-warning-snapshot.test.ts": "codex-integration", "codex-quota-auto-refresh-generation.test.ts": "codex-integration", + "pnpm-command-isolation.test.ts": "update", "provider-antigravity-quota-retry.test.ts": "providers", "project-config-warning-snapshot.test.ts": "codex-integration", "codex-quota-auto-refresh-generation.test.ts": "codex-integration", "codex-account-clear-paused.test.ts": "codex-integration", "responses-compaction-recovery.test.ts": "responses", "compaction-recovery-settings.test.ts": "config", "responses-compaction-recovery-policy.test.ts": "responses", "plugin-loader.test.ts": "lib", "plugin-upstream-hooks.test.ts": "lib", "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", diff --git a/src/cli/account-target.ts b/src/cli/account-target.ts index 00d875f7cf5..e0b433da7e1 100644 --- a/src/cli/account-target.ts +++ b/src/cli/account-target.ts @@ -15,7 +15,7 @@ const MAIN_ALIAS = "main"; export type CodexAccountTarget = | { id: string } - | { error: string; kind: "not_found" | "ambiguous" | "reserved" } + | { error: string; kind: "not_found" | "ambiguous" | "reserved" | "unavailable" } | { networkDown: true; transportError?: string }; /** Built-in selectors cannot be reused as Codex account aliases. */ @@ -35,18 +35,21 @@ export async function resolveCodexAccountTarget( requested: string, ): Promise { if (requested === MAIN_ALIAS || requested === MAIN_CODEX_ACCOUNT_ID) return { id: MAIN_CODEX_ACCOUNT_ID }; - if (isReservedCodexAccountWord(requested)) { - return { error: `"${requested}" is reserved; it clears the selection with \`ocx account use\` and names no account`, kind: "reserved" }; - } const res = await apiJson(deps, baseUrl, "GET", "/api/codex-auth/accounts"); if (res.status === 0) return { networkDown: true, transportError: res.transportError }; // The list is only needed to turn an alias into an id. If the proxy cannot produce it, send // the argument as the id it may already be and let the route answer, as the CLI always did. - if (res.status !== 200) return { id: requested }; + if (res.status !== 200) { + if (!isReservedCodexAccountWord(requested)) return { id: requested }; + return { error: `Cannot safely resolve reserved account selector "${requested}" while the account list is unavailable`, kind: "unavailable" }; + } const accounts = (Array.isArray(res.json.accounts) ? res.json.accounts : []) .filter((entry): entry is { id: string; alias?: unknown } => typeof entry === "object" && entry !== null && typeof (entry as { id?: unknown }).id === "string"); if (accounts.some(account => account.id === requested)) return { id: requested }; + if (isReservedCodexAccountWord(requested)) { + return { error: `"${requested}" is reserved; it clears the selection with \`ocx account use\` and names no account`, kind: "reserved" }; + } const exact = accounts.filter(account => account.alias === requested); const matches = exact.length > 0 ? exact @@ -69,10 +72,12 @@ export async function resolveCodexUseTarget( baseUrl: string, requested: string, ): Promise { - if (requested === AUTO_ACCOUNT_ARGUMENT) return { accountId: null }; const target = await resolveCodexAccountTarget(deps, baseUrl, requested); if ("networkDown" in target) return target; - if ("error" in target) return target; + if ("error" in target) { + if (requested === AUTO_ACCOUNT_ARGUMENT && target.kind === "reserved") return { accountId: null }; + return target; + } return { accountId: target.id }; } diff --git a/src/cli/account.ts b/src/cli/account.ts index 95bcd892eb7..025a1e3f7e1 100644 --- a/src/cli/account.ts +++ b/src/cli/account.ts @@ -45,6 +45,7 @@ const ACCOUNT_USAGE = `Usage: ocx account history openai [--limit <1-200>] [--json] ocx account current [--json] ocx account use [--json] + ocx account clear [--json] ocx account refresh [--json] ocx account auto-switch > [--json] ocx account alias [--json] @@ -68,8 +69,10 @@ const ACCOUNT_USAGE = `Usage: List and switch provider accounts and API-key pools (masked output only). 'main' selects the Codex App login for the openai account pool; 'auto' clears the -selection so the pool places work by its own strategy. A Codex account can be named -by the alias set with 'ocx account alias' wherever an id is accepted.`; +selection so the pool places work by its own strategy — unless an account actually +carries that id, which wins, so 'ocx account clear' is the spelling that always +clears. A Codex account can be named by the alias set with 'ocx account alias' +wherever an id is accepted.`; function consumeFlag(args: string[], flag: string): boolean { const idx = args.indexOf(flag); @@ -352,6 +355,45 @@ async function cmdUse(rest: string[], deps: AccountDeps): Promise { return 0; } +/** `ocx account clear` never resolves its argument as an account id, so an account literally + * named `auto` cannot shadow the verb that returns the pool to automatic selection. */ +async function cmdClear(rest: string[], deps: AccountDeps): Promise { + const wantsJson = consumeFlag(rest, "--json"); + const name = rest.shift(); + const leftover = leftoverArgsError(rest); + if (!name || leftover) { + if (leftover) console.error(leftover); + console.error(ACCOUNT_USAGE); + return 1; + } + const config = deps.loadConfigImpl?.() ?? loadConfig(); + const c = classifyAccount(config, name); + if ("error" in c) { + console.error(`Error: ${c.error}. Known candidates: ${candidateNames(config)}`); + return 1; + } + if (c.type !== "codex") { + console.error(`Error: ${name} has no automatic-selection pin to clear; clear applies to Codex account pools`); + return 1; + } + const baseUrl = await resolveBaseUrl(deps); + if (!baseUrl) return proxyUnreachable(); + const res = await apiJson(deps, baseUrl, "PUT", "/api/codex-auth/active", { accountId: null }); + if (res.status === 0) return proxyUnreachable(res.transportError); + if (res.status !== 200) return apiError(res.json, `failed to clear ${name}`, res.status); + const pinDrainReason = typeof res.json.pinDrainReason === "string" ? res.json.pinDrainReason : undefined; + if (wantsJson) { + console.log(JSON.stringify({ + ok: true, provider: name, type: c.type, activeId: null, + ...(pinDrainReason !== undefined ? { pinDrained: true, pinDrainReason } : {}), + }, null, 2)); + } else { + console.log(`${name}: automatic account selection (pin cleared)`); + } + await explainCodexUseOutcome(deps, baseUrl, name, null, pinDrainReason); + return 0; +} + export async function cmdAccount(args: string[], deps: AccountDeps = {}): Promise { const [sub, ...rest] = args; try { @@ -362,6 +404,7 @@ export async function cmdAccount(args: string[], deps: AccountDeps = {}): Promis } if (sub === "current") return await cmdCurrent(rest, deps); if (sub === "use") return await cmdUse(rest, deps); + if (sub === "clear") return await cmdClear(rest, deps); if (sub === "refresh") return await cmdRefresh(rest, deps); if (sub === "auto-switch") return await cmdAutoSwitch(rest, deps); if (sub === "alias" || sub === "rename") return await cmdAlias(rest, deps); diff --git a/src/codex/auth-api/routes.ts b/src/codex/auth-api/routes.ts index 5ad42d82f61..210dca82c04 100644 --- a/src/codex/auth-api/routes.ts +++ b/src/codex/auth-api/routes.ts @@ -213,7 +213,7 @@ export async function handleCodexAuthAPI( if (body.accountId === MAIN_CODEX_ACCOUNT_ID && hasLegacyMainCodexPoolAccount(runtimeConfig.codexAccounts)) { return jsonResponse({ error: "Remove the legacy __main__ pool row before selecting the Desktop account" }, 409); } - if (isCodexAccountPaused(runtimeConfig, targetAccountId)) { + if (body.accountId != null && isCodexAccountPaused(runtimeConfig, targetAccountId)) { return jsonResponse({ error: "Account is paused" }, 409); } if (body.accountId != null && body.accountId !== MAIN_CODEX_ACCOUNT_ID) { diff --git a/structure/providers/openai-accounts.md b/structure/providers/openai-accounts.md index 4415df5b377..5010979c458 100644 --- a/structure/providers/openai-accounts.md +++ b/structure/providers/openai-accounts.md @@ -174,8 +174,11 @@ Pool mode needs stable public names and a store that survives concurrent refresh - Public selectors are generated per account; the main login's selector is `main`, collision-suffixed if that name is taken, and it maps to the config-only sentinel `@main`, which sits outside the pool-account id grammar (`src/codex/account-namespaces.ts`, `src/codex/account-namespace-match.ts`). - Selectors must not collide with provider or combo ids. A user alias is display metadata; routing - consults credential identity, never the alias. + Selectors must not collide with provider or combo ids. The CLI's `auto` account control word is + resolved after exact stored account ids, so a legacy account with that id remains selectable + rather than invoking the control action; `ocx account clear` skips selector resolution entirely + and always clears the selection. A user alias is display metadata; routing consults + credential identity, never the alias. - The credential store is generation-guarded and refresh-locked (`src/codex/account-store.ts`): a refresh persists only if the generation it started from still holds, and a lost race raises a generation-conflict error instead of overwriting the newer credential. diff --git a/tests/cli/cli-account-alias-target.test.ts b/tests/cli/cli-account-alias-target.test.ts index 23a897d81a3..ba69f556691 100644 --- a/tests/cli/cli-account-alias-target.test.ts +++ b/tests/cli/cli-account-alias-target.test.ts @@ -23,16 +23,21 @@ interface Harness { listFailure: { status: number; error: string } | null; run: (args: string[]) => Promise<{ code: number; stdout: string; stderr: string }>; writes: () => Captured[]; + activeId: () => string | null; } function harness(providers: Record = {}): Harness { const requests: Captured[] = []; + // The mock active route persists what PUT writes so a verb that writes nothing (or the wrong + // thing) leaves observable state, not just a captured request. + let activeId: string | null = null; const state: Harness = { requests, accounts: [{ id: "chatgpt_1", plan: "pro", quota: null }], listFailure: null, run: async () => ({ code: 0, stdout: "", stderr: "" }), writes: () => requests.filter(r => r.method === "PUT" && r.path === "/api/codex-auth/active"), + activeId: () => activeId, }; const config = (): OcxConfig => ({ port: 10100, @@ -64,10 +69,10 @@ function harness(providers: Record = {}): Harness { return new Response(JSON.stringify({ accounts: state.accounts }), { status: 200 }); } if (captured.path === "/api/codex-auth/active") { - const pinned = captured.method === "PUT" - ? (captured.body as { accountId?: string | null } | undefined)?.accountId ?? null - : null; - return new Response(JSON.stringify({ ok: true, activeCodexAccountId: pinned, activeId: pinned }), { status: 200 }); + if (captured.method === "PUT") { + activeId = (captured.body as { accountId?: string | null } | undefined)?.accountId ?? null; + } + return new Response(JSON.stringify({ ok: true, activeCodexAccountId: activeId, activeId }), { status: 200 }); } return new Response(JSON.stringify({ ok: true }), { status: 200 }); }) as unknown as typeof fetch, @@ -166,14 +171,43 @@ describe("ocx account: alias and auto as account arguments", () => { expect(result.stdout).toContain("chatgpt_2"); }); - test("use auto clears the pin and says the pool decides from here", async () => { + test("use auto selects an exact account id before treating auto as the pool selector", async () => { const h = harness(); - const result = await h.run(["use", "openai", "auto"]); + const automatic = await h.run(["use", "openai", "auto"]); - expect(result.code).toBe(0); + expect(automatic.code).toBe(0); + expect(h.writes().at(-1)?.body).toEqual({ accountId: null }); + expect(automatic.stdout).toContain("automatic account selection"); + expect(automatic.stderr).not.toContain("may override this pin"); + + h.accounts.push({ id: "auto", plan: "pro", quota: null }); + const selected = await h.run(["use", "openai", "auto"]); + expect(selected.code).toBe(0); + expect(h.writes().at(-1)?.body).toEqual({ accountId: "auto" }); + expect(h.activeId()).toBe("auto"); + + const cleared = await h.run(["clear", "openai"]); + expect(cleared.code).toBe(0); + expect(h.activeId()).toBe(null); + + const paused = await h.run(["pause", "openai", "auto"]); + expect(paused.code).toBe(0); + expect(h.requests.filter(r => r.path === "/api/codex-auth/accounts/pause").at(-1)?.body) + .toEqual({ id: "auto", paused: true }); + }); + + test("an account named auto pins through use while clear still restores automatic selection", async () => { + const h = harness(); + h.accounts.push({ id: "auto", plan: "pro", quota: null }); + + const pinned = await h.run(["use", "openai", "auto"]); + expect(pinned.code).toBe(0); + expect(h.writes().at(-1)?.body).toEqual({ accountId: "auto" }); + + const cleared = await h.run(["clear", "openai"]); + expect(cleared.code).toBe(0); expect(h.writes().at(-1)?.body).toEqual({ accountId: null }); - expect(result.stdout).toContain("automatic account selection"); - expect(result.stderr).not.toContain("may override this pin"); + expect(cleared.stdout).toContain("automatic account selection"); }); test("missing and ambiguous aliases keep distinct errors before any write", async () => { @@ -220,8 +254,12 @@ describe("ocx account: alias and auto as account arguments", () => { const pause = await h.run(["pause", "openai", "auto"]); expect(pause.code).toBe(1); expect(pause.stderr).toContain("reserved"); - // The list only serves alias resolution: without it the argument is sent as an id, as before. h.listFailure = { status: 500, error: "list unavailable" }; + const automatic = await h.run(["use", "openai", "auto"]); + expect(automatic.code).toBe(1); + expect(automatic.stderr).toContain("Cannot safely resolve"); + expect(h.writes()).toHaveLength(0); + // The list only serves alias resolution: without it the argument is sent as an id, as before. const raw = await h.run(["use", "openai", "chatgpt_1"]); expect(raw.code).toBe(0); expect(h.writes().at(-1)?.body).toEqual({ accountId: "chatgpt_1" }); diff --git a/tests/codex-integration/codex-account-clear-paused.test.ts b/tests/codex-integration/codex-account-clear-paused.test.ts new file mode 100644 index 00000000000..ac971d97681 --- /dev/null +++ b/tests/codex-integration/codex-account-clear-paused.test.ts @@ -0,0 +1,55 @@ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { getDefaultConfig, loadConfig } from "../../src/config"; +import { handleCodexAuthAPI } from "../../src/codex/auth-api/routes"; +import { MAIN_CODEX_ACCOUNT_ID } from "../../src/codex/main-account"; +import { pinnedCodexAccountId, setCodexAccountPin } from "../../src/codex/account-priority"; +import { createTempHome } from "../helpers/temp-home"; + +let home: ReturnType; +beforeEach(() => { home = createTempHome("ocx-clear-paused-account-"); }); +afterEach(() => { home.remove(); }); + +const url = new URL("http://127.0.0.1/api/codex-auth/active"); +function request(accountId: string | null): Request { + return new Request(url, { method: "PUT", headers: { "content-type": "application/json" }, + body: JSON.stringify({ accountId }) }); +} + +test("null clears active selection and pin even when the main account is paused", async () => { + const config = getDefaultConfig(); + config.activeCodexAccountId = MAIN_CODEX_ACCOUNT_ID; + config.pausedCodexAccountIds = [MAIN_CODEX_ACCOUNT_ID]; + setCodexAccountPin(config, MAIN_CODEX_ACCOUNT_ID); + const response = await handleCodexAuthAPI(request(null), url, config); + expect(response?.status).toBe(200); + expect(await response!.json()).toMatchObject({ ok: true, activeCodexAccountId: null }); + expect(config.activeCodexAccountId).toBeUndefined(); + expect(pinnedCodexAccountId(config)).toBeUndefined(); + const stored = loadConfig(); + expect(stored.activeCodexAccountId).toBeUndefined(); + expect(pinnedCodexAccountId(stored)).toBeUndefined(); + expect(stored.pausedCodexAccountIds).toContain(MAIN_CODEX_ACCOUNT_ID); +}); + +test("clearing with paused main is idempotent and does not select it", async () => { + const config = getDefaultConfig(); + config.pausedCodexAccountIds = [MAIN_CODEX_ACCOUNT_ID]; + for (let i = 0; i < 2; i++) { + const response = await handleCodexAuthAPI(request(null), url, config); + expect(response?.status).toBe(200); + expect(config.activeCodexAccountId).toBeUndefined(); + expect(pinnedCodexAccountId(config)).toBeUndefined(); + } +}); + +test("an explicit paused main selection still fails without clearing an existing pin", async () => { + const config = getDefaultConfig(); + config.activeCodexAccountId = "pool-fixture"; + config.pausedCodexAccountIds = [MAIN_CODEX_ACCOUNT_ID]; + setCodexAccountPin(config, "pool-fixture"); + const response = await handleCodexAuthAPI(request(MAIN_CODEX_ACCOUNT_ID), url, config); + expect(response?.status).toBe(409); + expect(await response!.json()).toEqual({ error: "Account is paused" }); + expect(config.activeCodexAccountId).toBe("pool-fixture"); + expect(pinnedCodexAccountId(config)).toBe("pool-fixture"); +}); diff --git a/tests/codex-integration/codex-auth-api.test.ts b/tests/codex-integration/codex-auth-api.test.ts index 20d21eddb28..3ee450460f7 100644 --- a/tests/codex-integration/codex-auth-api.test.ts +++ b/tests/codex-integration/codex-auth-api.test.ts @@ -4280,20 +4280,6 @@ describe("codex-auth API", () => { expect(config.pausedCodexAccountIds).toEqual([MAIN_CODEX_ACCOUNT_ID]); }); - test("PUT /api/codex-auth/active rejects null when the effective main account is paused", async () => { - const config = makeConfig({ pausedCodexAccountIds: [MAIN_CODEX_ACCOUNT_ID] }); - const req = new Request("http://localhost/api/codex-auth/active", { - method: "PUT", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify({ accountId: null }), - }); - const resp = await handleCodexAuthAPI(req, new URL(req.url), config); - - expect(resp!.status).toBe(409); - expect(await resp!.json()).toEqual({ error: "Account is paused" }); - expect(config.activeCodexAccountId).toBeUndefined(); - }); - test("resuming restores eligibility and manual activation rejects paused accounts", async () => { const config = makeConfig({ codexAccounts: [{ id: "work", email: "work@example.test", isMain: false }], diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 04cfd1ace90..2924eaf5ca9 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -18,6 +18,7 @@ "restart-replacement.test.ts": "server", "deepseek-artifact-tool-schema.test.ts": "providers", "client-config-export-output-limit.test.ts": "config", + "codex-account-clear-paused.test.ts": "codex-integration", "openai-chat-serialized-tool-call-scaling.test.ts": "adapters/openai", "openai-chat-tool-call-id-remint.test.ts": "adapters/openai", "coding-agent-json-lines-scaling.test.ts": "providers", From 090e5e189ef06fcdb4a3bf66df1dc95d92f457d3 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 15:32:17 +0900 Subject: [PATCH 38/75] docs(devlog): record train 3 B3 build and evidence --- .../_plan/260927_merge_train_3/030_batch3.md | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/030_batch3.md b/devlog/_plan/260927_merge_train_3/030_batch3.md index 776d3526e5b..31a9ed95bbb 100644 --- a/devlog/_plan/260927_merge_train_3/030_batch3.md +++ b/devlog/_plan/260927_merge_train_3/030_batch3.md @@ -33,3 +33,30 @@ security reviews with no blocker recorded in this unit. - #5494: the regex is anchored to the token-plan phrasing, `/usage limit (?:has been )?reached|token-plan\s+\S+\s+quota has been exhausted/`, with a negative case for "quota exhausted for this minute". The hold is the existing ten-minute exhaustion cap, not the announced reset. + +## Build and evidence + +| Commit | What | +|---|---| +| `c79fe409c6` | #6049 (layout registries unioned) | +| `33b1920cce` | #6042 | +| `b697756cdb` | #6037 (`job.ts` import conflict: both imports kept; 1999 lines) | +| `8e08f26f86` | #6020 | +| `2a49525f1d` | #6020 review fix: generation-keyed retries, flat one-minute retry on `NativeMainBusyError`, three regression tests in `codex-quota-auto-refresh-generation.test.ts` (all three fail without the fix) | +| `9fdc0c4582` | #5494: token-plan exhaustion takes the ten-minute hold; positive and per-minute negative tests (the positive fails without the fix) | +| `cafe6202ad` | #6050 (the dev test pinning the old paused-main 409 on clear is removed; the new file covers the contract) | + +Kimi's note that #6050 regressed the `shadow` defaults and the Kiro-only `strategy` note came from diffing against +an older base; the squash onto current `dev` changes only the account-selection lines in the eight locales. + +Security receipts: #6042 dedicated review, BLOCKER no (connect-phase race stays documented, as the PR states). #6037 +review found no blocking defect; the updater launcher now trusts only root-owned absolute paths, and Ingwannu's +earlier CHANGES_REQUESTED findings (lexical ancestors, synchronous probes) are fixed at the carried head. + +Aside: #6037 still shows one CHANGES_REQUESTED review and #6020 two, both from earlier heads; this batch answers +#6020's findings in `2a49525f1d`. #5494's page shows the reporter's two messages and no maintainer reply; the fix +covers the part the repository can prove (the 60-second re-offer). Why the official DeepSeek stream ended early needs +the reporter's logs. + +Local proof at `cafe6202ad`: typecheck, structure and privacy exit 0; 13 focused files 619 pass, 3 skip, 0 fail; +combo failover files 297 pass; layout and ratchet guards 27 pass. From ac29ec5836719ab3cc5e10e6269334a88c192f50 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:05:31 +0900 Subject: [PATCH 39/75] docs(devlog): plan train 3 B4 --- devlog/_plan/260927_merge_train_3/040_batch4.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/040_batch4.md diff --git a/devlog/_plan/260927_merge_train_3/040_batch4.md b/devlog/_plan/260927_merge_train_3/040_batch4.md new file mode 100644 index 00000000000..ec3b86d4c00 --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/040_batch4.md @@ -0,0 +1,17 @@ +# B4 — native-main recovery fence + +Base: `dev` `bf6c57c0d7` (after B3 #6062). Branch `codex/train3-b4`. + +Previous D (B3): landed with exact-head CI; #6020's review findings and #5494 were fixed in the batch. Direction kept. + +| PR | Author | Plan | Kimi verdict | +|---|---|---|---| +| #6043 | luvs01 | Carry. Last-reference release no longer resets the native-main gate over a recovery fence that a profile transaction published for the same home. | LAND (no code change; carry onto current dev). The new release test fails on dev without the fix. | + +Held, with reasons: + +- #6044 (link relay authentication): security review recorded a blocker that the PR thread already lists as open; the + branch also conflicts with the relay rewrite in #6034/#5998. Not landable in this lane. +- #6051 (`.agents/skills` recipe): accurate and safe, but it creates a new skill root that neither `AGENTS.md` nor + the hygiene tests know about; the owner and author left that as a maintainer decision. +- #6027: the owner's three blockers are still open on the head. From f83cd2fc74727de052360504e9e2f30985771a7e Mon Sep 17 00:00:00 2001 From: Epinephrine Date: Sun, 27 Sep 2026 16:05:34 +0900 Subject: [PATCH 40/75] fix(codex): preserve transaction recovery fence on release (#6043) Carried from #6043 into merge train round 3. Co-authored-by: Epinephrine --- src/codex/native-profile-startup.ts | 22 ++- structure/codex-home.md | 13 +- .../native-profile-startup-release.test.ts | 135 +++++++++++++++++- 3 files changed, 158 insertions(+), 12 deletions(-) diff --git a/src/codex/native-profile-startup.ts b/src/codex/native-profile-startup.ts index 72bc6ace761..8d332787782 100644 --- a/src/codex/native-profile-startup.ts +++ b/src/codex/native-profile-startup.ts @@ -78,6 +78,8 @@ interface StartupEntry { recoveryStarted: boolean; policyBindingPending: boolean; settled: Promise; + /** Publication provenance only; never substitutes for the live convergence drain. */ + snapshotSettled?: Promise; resolveAcquisition?: (value: NativeMainStartupGateSnapshot) => void; deps: NativeMainStartupGateDeps; manager: NativeProfileManager; @@ -401,6 +403,7 @@ export function startNativeMainStartupLifecycle( released = true; entry!.refs = Math.max(0, entry!.refs - 1); if (entry!.refs !== 0) return; + const releasedEpoch = entry!.epoch; entry!.epoch += 1; entry!.sweepStopping = true; if (entry!.sweepTimer) clearTimeout(entry!.sweepTimer); @@ -413,11 +416,17 @@ export function startNativeMainStartupLifecycle( // the process exited, because a server whose config does not sync Codex installs a no-op // lifecycle that never touches the gate. // - // The gate state belonged only to this entry, so reset it to the process-initial state here, - // synchronously and before the first await: a NEW entry created for the same home afterwards - // re-arms its own gate and cannot be clobbered by this release. The epoch bump retires any - // in-flight `initializeNativeMainStartupGate`/convergence write from the released generation. - if (snapshot.homeId === homeId && !startupEntries.has(homeId)) { + // Reset only a snapshot published by this startup generation. A profile transaction can + // independently replace it with a same-home recovery fence, which must survive this release. + // The epoch provenance misses one case: that fence advances the global epoch while this + // generation's convergence is still in flight, and `completeNativeMainRecovery` then rebinds + // the shared `settled` to this entry's own pending chain without re-stamping either epoch. + // The pending snapshot is again this generation's own, so `settled` identity proves it too. + // Do this synchronously and before the first await: a NEW entry created for the same home + // afterwards re-arms its own gate and cannot be clobbered by this release. The epoch bump + // retires any in-flight convergence write from the released generation. + if (snapshot.homeId === homeId && !startupEntries.has(homeId) + && (epoch === releasedEpoch || settled === entry!.settled || settled === entry!.snapshotSettled)) { epoch += 1; snapshot = ready(null); settled = Promise.resolve(snapshot); @@ -728,6 +737,9 @@ export function completeNativeMainRecovery(homeId: string): boolean { clearAccountNeedsReauth(MAIN_CODEX_ACCOUNT_ID); snapshot = ready(homeId); settled = Promise.resolve(snapshot); + // Remember who published this snapshot without replacing an in-flight recovery/sweep + // promise: last-reference release must still drain that original convergence chain. + if (entry) entry.snapshotSettled = settled; return true; } diff --git a/structure/codex-home.md b/structure/codex-home.md index b7dfeb0860d..d0f7e4c5a15 100644 --- a/structure/codex-home.md +++ b/structure/codex-home.md @@ -141,11 +141,14 @@ subsystem from fencing native traffic or creating lock contention. Presence, an or any observation error still takes the locked sweep and fails closed; the fast path is based only on proven absence, never on an unreadable path. -Native-main admission is one process-global gate owned by the live startup entry. Releasing the -last reference to that entry returns the gate to its process-initial `ready` state synchronously, -before the release awaits anything, so a server stopped in the middle of startup convergence -cannot leave the process fenced for the servers that follow it; an entry created afterwards for -the same home arms its own gate, and the retired generation's late convergence writes are ignored. +Native-main admission is one process-global gate. Releasing the last reference to a startup entry +returns a snapshot published by that startup generation to the process-initial `ready` state +synchronously, before the release awaits anything, so a server stopped in the middle of startup +convergence cannot leave the process fenced for the servers that follow it. A same-home recovery +fence published independently by a profile transaction survives release. An entry created +afterwards for the same home arms its own gate, and the retired generation's late convergence +writes are ignored. +Recovery-completion provenance is separate from the convergence promise: release retains the owner through the pending recovery and its following stage sweep. `tests/codex-integration/native-profile-startup-release.test.ts` pins that ordering. A sibling instance — `ocx start --port ` while a live proxy serves the configured port, the diff --git a/tests/codex-integration/native-profile-startup-release.test.ts b/tests/codex-integration/native-profile-startup-release.test.ts index 100f0de18ca..3e729db0ed1 100644 --- a/tests/codex-integration/native-profile-startup-release.test.ts +++ b/tests/codex-integration/native-profile-startup-release.test.ts @@ -1,8 +1,9 @@ import { afterEach, describe, expect, test } from "bun:test"; -import { mkdirSync, mkdtempSync, realpathSync } from "node:fs"; +import { mkdirSync, mkdtempSync, realpathSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { nativeMainOwnerSnapshot } from "../../src/codex/native-main-owner"; import { withNativeMainExclusiveClaim } from "../../src/codex/native-main-claim"; import { NativeProfileManager } from "../../src/codex/native-profile-manager"; import type { @@ -10,6 +11,7 @@ import type { NativeProfileRecoveryState, } from "../../src/codex/native-profile-store"; import { + blockNativeMainRecovery, completeNativeMainRecovery, flushNativeMainStartupReleases, isNativeMainTrafficBlocked, @@ -132,6 +134,136 @@ function startLifecycle(deps: NativeMainStartupGateDeps): NativeMainStartupLifec } describe("a released native-main startup entry cannot leave the process fenced", () => { + test("hard-lock-off recovery completion keeps the convergence drain and owner until the sweep ends", async () => { + const f = fabricatedHome("complete-before-release-drain"); + writeFileSync(join(process.env.OPENCODEX_HOME!, "config.json"), JSON.stringify({ codexMainAccountHardLock: false })); + const recovery = barrier(), recoveryEntered = barrier(), sweep = barrier(), sweepEntered = barrier(); + let state: NativeProfileRecoveryState = "journal"; + const lifecycle = startLifecycle({ + manager: fabricatedManager(f.context, { sweepStages: async () => { + sweepEntered.open(); await sweep.promise; return { plaintextMayRemain: false }; + } }), + probeRecoveryState: () => state, + beforeRecovery: async () => { recoveryEntered.open(); await recovery.promise; state = "none"; }, + owner: OWNER, + }); + let flight: Promise | undefined; + let released = false; + try { + await within(recoveryEntered.promise, "recovery entry"); + const convergence = lifecycle.settled; + expect(blockNativeMainRecovery(f.homeId, "manual")).toBe(true); + expect(completeNativeMainRecovery(f.homeId)).toBe(true); + expect(lifecycle.settled).toBe(convergence); + flight = lifecycle.release().then(() => { released = true; }); + expect(nativeMainStartupGateSnapshot()).toEqual({ status: "ready", homeId: null }); + expect(nativeMainOwnerSnapshot(f.context)?.status).toBe("held"); + recovery.open(); + await within(sweepEntered.promise, "the post-recovery sweep"); + expect(released).toBe(false); + expect(nativeMainOwnerSnapshot(f.context)?.status).toBe("held"); + sweep.open(); + await within(flight, "release after the sweep"); + expect(released).toBe(true); + expect(nativeMainOwnerSnapshot(f.context)).toBeNull(); + } finally { + recovery.open(); sweep.open(); + await within(flight ?? lifecycle.release(), "final release"); + } + }); + + test("a release preserves a recovery fence published after startup", async () => { + const f = fabricatedHome("transaction-recovery-home"); + const lifecycle = startLifecycle({ manager: f.manager, probeRecoveryState: () => "none", owner: OWNER }); + + expect(await within(lifecycle.settled, "startup convergence")).toEqual({ + status: "ready", + homeId: f.homeId, + }); + expect(blockNativeMainRecovery(f.homeId, "manual")).toBe(true); + + await within(lifecycle.release(), "the lifecycle release"); + expect(nativeMainStartupGateSnapshot()).toEqual({ + status: "blocked", + homeId: f.homeId, + reason: "manual-recovery", + }); + expect(isNativeMainTrafficBlocked()).toBe(true); + }); + + test("a release after a completed transaction recovery still opens the gate", async () => { + const f = fabricatedHome("completed-recovery-fence-home"); + const recovery = barrier(); + const recoveryEntered = barrier(); + let recoveryState: NativeProfileRecoveryState = "journal"; + const lifecycle = startLifecycle({ + manager: f.manager, + probeRecoveryState: () => recoveryState, + beforeRecovery: async () => { recoveryEntered.open(); await recovery.promise; recoveryState = "none"; }, + owner: OWNER, + }); + let flight: Promise | undefined; + try { + await within(recoveryEntered.promise, "the owned recovery phase to start"); + // A profile transaction fences the home while startup convergence is still in flight, + // advancing the global epoch past the entry's own. + expect(blockNativeMainRecovery(f.homeId, "manual")).toBe(true); + // Completing it re-arms the pending entry: the gate content is again the startup + // generation's own recovery-pending snapshot, just under an epoch the entry predates. + expect(completeNativeMainRecovery(f.homeId)).toBe(true); + expect(nativeMainStartupGateSnapshot()).toEqual({ + status: "blocked", + homeId: f.homeId, + reason: "recovery-pending", + }); + + flight = lifecycle.release(); + expect(nativeMainStartupGateSnapshot()).toEqual({ status: "ready", homeId: null }); + expect(isNativeMainTrafficBlocked()).toBe(false); + + recovery.open(); + await within(flight, "the release flight to settle"); + expect(nativeMainStartupGateSnapshot()).toEqual({ status: "ready", homeId: null }); + expect(isNativeMainTrafficBlocked()).toBe(false); + } finally { + recovery.open(); + await within(flight ?? lifecycle.release(), "the release flight to settle"); + } + }); + + test("a release orphans no sweep-published cleanup fence after a completed recovery", async () => { + const f = fabricatedHome("sweep-fence-home"); + // Hard lock off: completeNativeMainRecovery takes the direct ready(homeId) path, which used + // to leave the live entry's provenance stale while its own stage sweep republished the fence. + writeFileSync(join(process.env.OPENCODEX_HOME!, "config.json"), JSON.stringify({ codexMainAccountHardLock: false })); + const lifecycle = startLifecycle({ + manager: fabricatedManager(f.context, { sweepStages: async () => ({ plaintextMayRemain: true }) }), + probeRecoveryState: () => "none", + owner: OWNER, + stageSweepIntervalMs: 10, + }); + // Convergence ends fenced: the sweep always finds plaintext residue. + expect(await within(lifecycle.settled, "startup convergence")).toMatchObject({ status: "blocked" }); + + // The transaction fence + completion advance the global epoch past the entry's provenance. + expect(blockNativeMainRecovery(f.homeId, "manual")).toBe(true); + expect(completeNativeMainRecovery(f.homeId)).toBe(true); + expect(nativeMainStartupGateSnapshot()).toEqual({ status: "ready", homeId: f.homeId }); + + // The entry's next scheduled sweep republishes its cleanup fence under that provenance. + const deadline = Date.now() + 5_000; + while (nativeMainStartupGateSnapshot().status !== "blocked" && Date.now() < deadline) { + await new Promise(resolve => setTimeout(resolve, 10)); + } + expect(nativeMainStartupGateSnapshot()).toEqual({ + status: "blocked", homeId: f.homeId, reason: "stage-cleanup-required", + }); + + await within(lifecycle.release(), "the lifecycle release"); + expect(nativeMainStartupGateSnapshot()).toEqual({ status: "ready", homeId: null }); + expect(isNativeMainTrafficBlocked()).toBe(false); + }); + test("a release during recovery resets the gate and ignores the convergence that follows", async () => { const f = fabricatedHome("release-during-recovery-home"); const recovery = barrier(); @@ -342,4 +474,3 @@ describe("a released native-main startup entry cannot leave the process fenced", } }); }); - From 3fc74c06277bf955237047f7902f854e023df088 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:06:07 +0900 Subject: [PATCH 41/75] docs(devlog): record train 3 B4 evidence --- devlog/_plan/260927_merge_train_3/040_batch4.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/040_batch4.md b/devlog/_plan/260927_merge_train_3/040_batch4.md index ec3b86d4c00..8302cd763f5 100644 --- a/devlog/_plan/260927_merge_train_3/040_batch4.md +++ b/devlog/_plan/260927_merge_train_3/040_batch4.md @@ -15,3 +15,12 @@ Held, with reasons: - #6051 (`.agents/skills` recipe): accurate and safe, but it creates a new skill root that neither `AGENTS.md` nor the hygiene tests know about; the owner and author left that as a maintainer decision. - #6027: the owner's three blockers are still open on the head. + +## Build and evidence + +`f83cd2fc74` carries #6043 onto `bf6c57c0d7` (3 files; the GUI test edits from its earlier head are gone). Kimi's +negative control on current dev: "a release preserves a recovery fence published after startup" fails without the +fix. Local: `native-profile-startup-release.test.ts` 8 pass; typecheck, structure and privacy exit 0. + +Aside: #6043, #6044 and #6051 pages captured in `.tmp/aside/`. #6044's thread shows the author agreeing the relay +finding is still open, which matches the security review held in scratch. From 3a5c3a3c22e20a1908b99b50e039eff6b4420911 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:26:01 +0900 Subject: [PATCH 42/75] docs(devlog): plan train 3 B5 --- devlog/_plan/260927_merge_train_3/050_batch5.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/050_batch5.md diff --git a/devlog/_plan/260927_merge_train_3/050_batch5.md b/devlog/_plan/260927_merge_train_3/050_batch5.md new file mode 100644 index 00000000000..533f71915d5 --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/050_batch5.md @@ -0,0 +1,17 @@ +# B5 — narrowed carries, a Command Code retry default, and evidence-backed closes + +Base: `dev` `4b3737fc5c` (after B4 #6063). Branch `codex/train3-b5`. + +Previous D (B4): #6043 landed; #6044 held on its security blocker, #6051 on the owner's `.agents/` decision. The +coordinator asked the lane to continue until nothing in scope is landable. + +| Item | Plan | +|---|---| +| #5953 (codingbooo) → #5465 | Carry, then narrow `protectGlmSummaryBudget` as the maintainer round asked: Z.AI host only (from the provider base URL), caller effort `high`/`max` only, the checkpoint shape (summary instruction plus a `` transcript of at least 2000 characters), and each tiny cap field (≤1024) raised on its own to 8192. Negative tests for each boundary. | +| #5180 (issue, found through Aside) | With no `retryOn429` knob, a key-auth Command Code destination gets the patient same-key policy OpenCode Go already has, so a burst 429 on a long turn waits (honoring Retry-After) instead of failing to the client. An explicit `retryOn429`, including `enabled: false`, still wins; OAuth is never replayed. | +| #6027 (codingbooo) → #5569 | Carry, then fix the owner's three blockers: replace only when the whole body carries exactly one `` block (otherwise pass through); store a new snapshot only after `prepareResponsesRequest` reaches its success return, so a rejected first request pins nothing; share a snapshot without a principal only for loopback admission. Tests for each. | +| #4055 | Close as fixed for the reported Tailscale Serve case (12-hour identity sessions with sliding renewal, #2776), and correct the stale `management-api.md` sentence that says remote binds never get a session. | +| #3433 | Close with evidence: per-conversation identifiers are preserved at the forward boundary (#4365) and the managed Hermes export now sends one (#5742); 26 pinned tests pass. | + +Held: #6030 (draft; launchd PATH adoption drops non-PATH changes, two ratchet breaches, WinSW gap, conflict), +#4143 (needs the reporter's desktop routing details). From 79db34972fb3301f26bd6023de1a1c6207c5abae Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:30:32 +0900 Subject: [PATCH 43/75] docs(devlog): fold the B5 audit --- devlog/_plan/260927_merge_train_3/050_batch5.md | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/050_batch5.md b/devlog/_plan/260927_merge_train_3/050_batch5.md index 533f71915d5..04e8c990816 100644 --- a/devlog/_plan/260927_merge_train_3/050_batch5.md +++ b/devlog/_plan/260927_merge_train_3/050_batch5.md @@ -15,3 +15,14 @@ coordinator asked the lane to continue until nothing in scope is landable. Held: #6030 (draft; launchd PATH adoption drops non-PATH changes, two ratchet breaches, WinSW gap, conflict), #4143 (needs the reporter's desktop routing details). + +## Audit (Kimi, NEAR-PASS) and folded decisions + +- #5180: the predicate is key auth plus the existing `isCanonicalCommandCodeBaseUrl`; a row repointed at a custom + relay keeps fail-fast unless `retryOn429` is set. +- #6027: `snapshotSkillsCatalogInBody` splits into a replace-only lookup before parsing and a store call at the + success return. Residuals named in the PR: a snapshot can be pinned by a turn whose upstream request later fails; + on a no-auth server bound beyond loopback, snapshots are keyed by conversation id alone, matching that server's + trust model. +- #5953: the gate reads effective effort after combo overrides; transcript shapes that #5465 does not show stay + unprotected, which is today's behavior. From 4fda15d3383e8d5f27bcff00ca217926184f2894 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:31:12 +0900 Subject: [PATCH 44/75] fix(command-code): wait out a burst 429 on the canonical key endpoint Fixes #5180. Without a retryOn429 knob, a single Command Code key failed a long muse-spark turn on the first 429, because a single key cannot fail over and the Codex client does not retry 429. Key-auth rows at the canonical Command Code endpoints now get the patient same-key policy OpenCode Go already has (6 replays, 10 s interval, 60 s cap, Retry-After honored). OAuth rows are never replayed, an explicit retryOn429 including enabled:false still wins, and a custom relay keeps fail-fast. --- .../docs/reference/configuration/providers.md | 2 +- src/providers/key-failover.ts | 18 ++++++++++----- structure/transports/streaming-health.md | 2 +- tests/providers/rate-limit-retry.test.ts | 22 +++++++++++++++++++ 4 files changed, 37 insertions(+), 7 deletions(-) diff --git a/docs-site/src/content/docs/reference/configuration/providers.md b/docs-site/src/content/docs/reference/configuration/providers.md index c66261b0425..a440ec05c02 100644 --- a/docs-site/src/content/docs/reference/configuration/providers.md +++ b/docs-site/src/content/docs/reference/configuration/providers.md @@ -267,7 +267,7 @@ Providers can expose a built-in shorthand, such as `agy` for `google-antigravity | `responsesItemIdRepair?` | `{ message?: string[]; reasoning?: string[]; repairMissingTerminalIds?: boolean; repairInvalidIds?: boolean }` | Disabled-by-default downstream SSE repair for exact placeholder ids, missing terminal ids, and (with `repairInvalidIds`) message/reasoning ids missing the canonical `msg_`/`rs_` prefix. Function-call ids are never rewritten. Built-in DeepSeek enables the last two by default. | | `responsesSnapshotRepair?` | `boolean` | Disabled-by-default client-facing repair for sparse Responses lifecycle snapshots in SSE and JSON. Fills missing canonical status, output, and tool metadata while raw inspection and persistence remain unchanged. | | `webSearchBridge?` | `{ enabled?: boolean; backend?: "ollama" \| "openai" \| "anthropic" \| "xai" \| "gemini" \| "exa"; maxSearches?: number; timeoutMs?: number; endpoint?: string }` | Key-auth `openai-responses` passthrough providers only. Off by default. Codex always declares the hosted `web_search` tool, and the passthrough relays it on the assumption the destination executes it. A gateway that does not run hosted search answers with a `function_call` named `web_search` that nothing runs, and the undeclared-tool guard ends the turn. With `enabled: true` and an explicit `backend` OpenCodex intercepts that call, runs the search itself, feeds the result back to the same upstream, and shows Codex a hosted `web_search_call` cell. Never armed for `authMode: "forward"` (ChatGPT already searches) or for a provider that executes hosted search upstream. `backend` is required; there is no implicit default and a missing credential for the named backend leaves the bridge disarmed rather than falling through to another paid search. `ollama` reuses this provider's own API key on `POST /api/web_search`, so the origin must be `https://ollama.com` unless the operator names `endpoint` explicitly. `openai` / `anthropic` / `xai` / `gemini` / `exa` reuse the matching sidecar executor and that executor's own credential (`webSearchSidecar.exaApiKey` for Exa). The search model comes from `webSearchSidecar.model` only when `webSearchSidecar.backend` resolves to the same backend this bridge names; otherwise the bridge runs that backend's own default, because a model chosen for one vendor is rejected by another. An unset `webSearchSidecar.backend` resolves to `openai`, so an unset-backend model reaches an `openai` bridge and no other. There is no per-provider bridge model override. Streaming turns only. A turn that mixes `web_search` with another client tool call still fails closed rather than dropping the client's call. Assistant text such as XML-like `` prose is not executed. Defaults: `maxSearches: 3` (1..10), `timeoutMs: 60000` (1000..600000). | -| `retryOn429?` | `{ enabled?: boolean; attempts?: number; intervalMs?: number; maxIntervalMs?: number; respectRetryAfter?: boolean }` | API-key providers only (`authMode: "key"`). Opt-in same-target 429 retry: when `retryOn429` is absent the feature is off; object presence enables it unless `enabled: false`. On 429 the proxy waits (upstream `Retry-After` or the fixed interval) and replays the identical request on the same key before any key failover — across the main text-turn recovery loop, the Responses passthrough wire, the image/video bridge, the web-search sidecar, and terminal continuations. Only pre-stream HTTP 429 responses are eligible for replay; custom `runTurn` transports are outside the HTTP retry loop. `attempts` counts same-key replays after the first 429 (total sends = `attempts` + 1) and is one request-wide budget shared by the main recovery loop, the terminal-guard continuation, and bridge retries. Exhausting `attempts` only stops further same-key replays: normal key failover or final-error handling then applies per the available targets — on the key-auth passthrough wire there is no failover, so the exhausted 429 surfaces as-is. Codex itself never retries 429, so this is the only defense for single-key providers. Defaults: `enabled: true`, `attempts: 3`, `intervalMs: 5000`, `maxIntervalMs: 60000` (any single wait is capped at `maxIntervalMs`, itself capped at 600000), `respectRetryAfter: true`. | +| `retryOn429?` | `{ enabled?: boolean; attempts?: number; intervalMs?: number; maxIntervalMs?: number; respectRetryAfter?: boolean }` | API-key providers only (`authMode: "key"`). Opt-in same-target 429 retry: when `retryOn429` is absent the feature is off, except that key-auth OpenCode Go and Command Code (canonical endpoints) fall back to a patient policy (6 replays, 10 s interval, 60 s cap); object presence enables it unless `enabled: false`, which also turns that fallback off. On 429 the proxy waits (upstream `Retry-After` or the fixed interval) and replays the identical request on the same key before any key failover — across the main text-turn recovery loop, the Responses passthrough wire, the image/video bridge, the web-search sidecar, and terminal continuations. Only pre-stream HTTP 429 responses are eligible for replay; custom `runTurn` transports are outside the HTTP retry loop. `attempts` counts same-key replays after the first 429 (total sends = `attempts` + 1) and is one request-wide budget shared by the main recovery loop, the terminal-guard continuation, and bridge retries. Exhausting `attempts` only stops further same-key replays: normal key failover or final-error handling then applies per the available targets — on the key-auth passthrough wire there is no failover, so the exhausted 429 surfaces as-is. Codex itself never retries 429, so this is the only defense for single-key providers. Defaults: `enabled: true`, `attempts: 3`, `intervalMs: 5000`, `maxIntervalMs: 60000` (any single wait is capped at `maxIntervalMs`, itself capped at 600000), `respectRetryAfter: true`. | | `transientRetryOn5xx?` | `{ enabled?: boolean; attempts?: number }` | Key-auth `openai-chat` and `openai-responses` providers only. `authMode: "forward"` providers (the ChatGPT account pool) never read this option and keep the default ladder. Opt-in retry for pre-stream transient upstream statuses (500, 502, 503, 504, 520, 521, 522): absent means off, object presence enables it unless `enabled: false`. Covers the initial Responses request, the Responses passthrough lane and each of its recovery legs (OAuth-401 replay, same-target 429 replay, validated rebuild), the terminal-guard continuation, and native `/v1/chat/completions`. `attempts` is the TOTAL number of upstream sends allowed for one request including the first (1..10, default 3) — it is one budget shared with connection-reset recovery, so `3` means at most three real requests reach the provider. On the Responses passthrough lane the configured value is additionally intersected with the request-wide send allowance, so a value below that allowance narrows the ladder exactly while a value above it does not raise the bound. Waits use a fixed 400 ms exponential backoff capped at 5 s and honor `Retry-After`. Separate from `retryOn429`, which handles rate limiting; mid-stream failures are never replayed. | | `retryOnReset?` | `{ enabled?: boolean; replacements?: number }` | Native `openai-responses` providers, including `authMode: "forward"`. Opt-in replacement of a send that failed while the caller had observed nothing: absent means off, object presence enables it unless `enabled: false`. Covers both ambiguous stages — a connection that died before any response header, and an SSE body that died after the header while carrying only control events. A canonical ChatGPT upstream WebSocket that closed or errored after its create frame left, before any Responses event, is covered the same way, and its replacement is sent over HTTP. Only a self-contained request is ever replaced: `store: false`, complete `input`, no `previous_response_id`, `conversation` or `stream_id`, and only client-executed tools. `replacements` is the number of replacement sends ONE logical request may make across every leg and every combo child (1..2, default 1) — not a per-leg retry count and not a send budget, so a replacement still has to fit inside the send allowance the leg already had. A request that already emitted output or a tool call is never replaced, whatever this is set to. The replacement inference may still be billed if the origin had already started the first one, which is why this is off by default. | | `autoToolChoiceOnlyModels?` | `string[]` | Models whose `tool_choice` accepts only `auto` or `none`; forced choices are downgraded. | diff --git a/src/providers/key-failover.ts b/src/providers/key-failover.ts index c96b555029b..cbaca42739c 100644 --- a/src/providers/key-failover.ts +++ b/src/providers/key-failover.ts @@ -12,7 +12,7 @@ import { commitProviderApiKeySelection } from "./api-key-selection"; import type { ProviderApiKeySelection } from "../types/provider"; import { routedProviderConfig } from "../router"; import { getProviderRegistryEntry } from "./registry"; -import { normalizedBaseUrl } from "./quota/vendor-probes-key"; +import { isCanonicalCommandCodeBaseUrl, normalizedBaseUrl } from "./quota/vendor-probes-key"; import type { OcxConfig, OcxProviderConfig, RateLimitRetryPolicy, ResetReplayPolicy, TransientRetryPolicy } from "../types"; import { OPENCODE_GO_SESSION_HEADER } from "./opencode-go-transport"; import { resolveProviderTransport, type OcxProviderTransport } from "./xai-transport"; @@ -566,7 +566,8 @@ export function selectProactiveApiKeyTransport( * `enabled: false` to opt out). When the knob is absent, the OpenCode Go destination * (subscription traffic such as Muse Spark) falls back to a patient same-key policy so a * burst 429 waits and replays instead of surfacing to the client and aborting a long - * session; every other provider without the knob keeps today's fail-fast behavior. + * session; key-auth Command Code at its canonical endpoints gets the same fallback (#5180), and + * every other provider without the knob keeps today's fail-fast behavior. * OAuth/forward/local credentials are never replayed on the same token. The returned * policy is fully defaulted so callers never re-check fields. */ @@ -590,10 +591,17 @@ export function rateLimitRetryPolicyFor( respectRetryAfter: policy.respectRetryAfter ?? DEFAULT_RATE_LIMIT_RETRY.respectRetryAfter, }; } - // No explicit knob: patient fallback for the OpenCode Go destination only. + // No explicit knob: patient fallback for OpenCode Go and canonical Command Code only. if (provider.authMode !== undefined && provider.authMode !== "key") return null; - if (!isOpenCodeGoDestination(provider)) return null; - return { ...OPENCODE_GO_RATE_LIMIT_RETRY }; + if (isOpenCodeGoDestination(provider)) return { ...OPENCODE_GO_RATE_LIMIT_RETRY }; + // Command Code's Provider API rate-limits long muse-spark turns the same way (#5180): a single + // key cannot fail over, and the Codex client does not retry a 429, so the turn aborts. The key + // endpoint gets the same patient same-key policy. A row repointed at a custom relay is not the + // canonical endpoint and keeps fail-fast unless it sets `retryOn429`. + if (typeof provider.baseUrl === "string" && isCanonicalCommandCodeBaseUrl(provider.baseUrl.trim())) { + return { ...OPENCODE_GO_RATE_LIMIT_RETRY }; + } + return null; } /** diff --git a/structure/transports/streaming-health.md b/structure/transports/streaming-health.md index b87bd6444eb..3e01e3894d2 100644 --- a/structure/transports/streaming-health.md +++ b/structure/transports/streaming-health.md @@ -188,7 +188,7 @@ once the server observes the client disconnect (Bun propagates it asynchronously cancelled with 499 before any replay; because the propagation is async, a replay may precede the cancel if the interval elapses first (bounded by the same `attempts` budget). -OpenCode Go (`https://opencode.ai/zen/go/v1`, serving subscription traffic such as Muse Spark) ships a patient same-target fallback when no explicit `retryOn429` is configured: same-key wait-and-replay with a 10s interval and a 60s cap, `Retry-After` honored. Replays draw from the shared per-request send budget, so a burst typically absorbs a couple of paced sends before the 429 surfaces — without this, a single-key pool surfaced the first 429 immediately and the client’s own retry budget aborted the goal (`exceeded retry limit, last status: 429`). An explicit `retryOn429` — including `enabled: false` — always overrides the fallback; every other provider without the knob keeps fail-fast behavior. +OpenCode Go (`https://opencode.ai/zen/go/v1`, serving subscription traffic such as Muse Spark) ships a patient same-target fallback when no explicit `retryOn429` is configured: same-key wait-and-replay with a 10s interval and a 60s cap, `Retry-After` honored. Replays draw from the shared per-request send budget, so a burst typically absorbs a couple of paced sends before the 429 surfaces — without this, a single-key pool surfaced the first 429 immediately and the client’s own retry budget aborted the goal (`exceeded retry limit, last status: 429`). The same fallback covers key-auth Command Code at its canonical endpoints (`https://api.commandcode.ai/provider/v1` and the API root), where long muse-spark turns hit the same burst limit (#5180); OAuth rows are never replayed on the same token, and a row repointed at a custom relay keeps fail-fast. An explicit `retryOn429` — including `enabled: false` — always overrides the fallback; every other provider without the knob keeps fail-fast behavior. Provider-level `requestPacing` is the proactive companion to `retryOn429`. It reserves outbound request-start slots before transport work begins, so a known RPM ceiling does not have to fail once diff --git a/tests/providers/rate-limit-retry.test.ts b/tests/providers/rate-limit-retry.test.ts index e028f6abc67..adc82ad669f 100644 --- a/tests/providers/rate-limit-retry.test.ts +++ b/tests/providers/rate-limit-retry.test.ts @@ -109,6 +109,28 @@ describe("rateLimitRetryPolicyFor", () => { } }); + // #5180: a single Command Code key cannot fail over and the Codex client does not retry a 429, + // so a burst on a long muse-spark turn aborted it. Both canonical endpoints wait patiently. + test("falls back to the patient policy for key-auth Command Code without the knob", () => { + const patient = { enabled: true, attempts: 6, intervalMs: 10_000, maxIntervalMs: 60_000, respectRetryAfter: true }; + for (const baseUrl of ["https://api.commandcode.ai/provider/v1", "https://api.commandcode.ai"]) { + expect(rateLimitRetryPolicyFor({ baseUrl, authMode: "key", adapter: "openai-chat" } as OcxProviderConfig)) + .toEqual(patient); + expect(rateLimitRetryPolicyFor({ baseUrl, adapter: "openai-chat" } as OcxProviderConfig)).toEqual(patient); + } + // OAuth Command Code is never replayed on the same token. + expect(rateLimitRetryPolicyFor({ + baseUrl: "https://api.commandcode.ai", authMode: "oauth", + } as OcxProviderConfig)).toBeNull(); + // An explicit opt-out wins, and a custom relay keeps fail-fast. + expect(rateLimitRetryPolicyFor({ + baseUrl: "https://api.commandcode.ai/provider/v1", retryOn429: { enabled: false }, + } as OcxProviderConfig)).toBeNull(); + expect(rateLimitRetryPolicyFor({ + baseUrl: "https://relay.example.test/provider/v1", authMode: "key", + } as OcxProviderConfig)).toBeNull(); + }); + test("honors explicit values", () => { expect(rateLimitRetryPolicyFor({ retryOn429: { attempts: 10, intervalMs: 1_000, maxIntervalMs: 5_000, respectRetryAfter: false }, From 3876151d017718d9d02f172e1073e9a4d2f9779b Mon Sep 17 00:00:00 2001 From: codingbo Date: Sun, 27 Sep 2026 16:31:17 +0900 Subject: [PATCH 45/75] fix(adapters): protect tiny standalone GLM summary compaction from reasoning exhaustion (#5953) Carried from #5953 into merge train round 3. Co-authored-by: codingbo --- src/adapters/openai-chat.ts | 14 ++-- src/adapters/openai-chat/passthrough.ts | 2 + src/adapters/openai-chat/summary-budget.ts | 46 +++++++++++++ .../openai/openai-chat-glm-summary.test.ts | 65 +++++++++++++++++++ 4 files changed, 118 insertions(+), 9 deletions(-) create mode 100644 src/adapters/openai-chat/summary-budget.ts create mode 100644 tests/adapters/openai/openai-chat-glm-summary.test.ts diff --git a/src/adapters/openai-chat.ts b/src/adapters/openai-chat.ts index cd2e12c6183..85dbe21c569 100644 --- a/src/adapters/openai-chat.ts +++ b/src/adapters/openai-chat.ts @@ -1,11 +1,12 @@ import { hasShrinkableOpenAIChatImages, normalizeOpenAIChatImages } from "./openai-chat-images"; +import { protectGlmSummaryBudget, resolveMaxTokens } from "./openai-chat/summary-budget"; import { chatParallelToolCallsWireValue } from "./openai-chat/parallel-tool-calls"; import { applyExplicitChatReasoningWirePolicy } from "./openai-chat/reasoning-wire"; import type { AdapterRequest, IncomingMeta, ProviderAdapter } from "./base"; import type { AdapterEvent, OcxParsedRequest, OcxProviderConfig, OcxUsage } from "../types"; import { modelInList } from "../types"; import { createInlineThinkContentSplitter, splitInlineThinkContent } from "./inline-think-tags"; -import { mapReasoningEffort, modelRecordValue } from "../reasoning-effort"; +import { mapReasoningEffort } from "../reasoning-effort"; import { debugProviderDiagnostic } from "../lib/debug"; import { sseFieldValue } from "../lib/sse-decoder"; import { isDebugEnabled } from "../lib/debug-settings"; @@ -51,12 +52,6 @@ export { stripBracketedModelSuffix } from "./openai-chat/wire"; export { buildOpenAIChatPassthroughRequest } from "./openai-chat/passthrough"; export { formatOpenAIChatErrorBody } from "./openai-chat/errors"; -function resolveMaxTokens(provider: OcxProviderConfig, parsed: OcxParsedRequest): number | undefined { - return parsed.options.maxOutputTokens - ?? modelRecordValue(provider.modelMaxOutputTokens, parsed.modelId) - ?? provider.defaultMaxOutputTokens; -} - function thinkingBudgetForEffort(parsed: OcxParsedRequest, reasoningEffort: string, maxOutputTokens?: number): number | undefined { if (parsed.options.reasoning === "minimal") return 0; const maxBudget = maxOutputTokens ?? 32768; @@ -150,12 +145,13 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd body.stop = parsed.options.stopSequences; } const reasoningDisabled = modelInList(provider.noReasoningModels, parsed.modelId); - const reasoningEffort = mapReasoningEffort(provider, parsed.modelId, parsed.options.reasoning); + const requestedEffort = protectGlmSummaryBudget(body) ? "low" : parsed.options.reasoning; + const reasoningEffort = mapReasoningEffort(provider, parsed.modelId, requestedEffort); const explicitReasoning = applyExplicitChatReasoningWirePolicy({ provider, modelId: parsed.modelId, hasTools: !!tools, - requestedEffort: parsed.options.reasoning, + requestedEffort, wireEffort: reasoningEffort, reasoningDisabled, body, diff --git a/src/adapters/openai-chat/passthrough.ts b/src/adapters/openai-chat/passthrough.ts index cedc86f8c46..9350ada953e 100644 --- a/src/adapters/openai-chat/passthrough.ts +++ b/src/adapters/openai-chat/passthrough.ts @@ -1,3 +1,4 @@ +import { protectGlmSummaryBudget } from "./summary-budget"; import { openAIChatTransport, stripBracketedModelSuffix } from "./wire"; import type { AdapterRequest } from "../base"; import { frameAgentRouterMessages } from "../agentrouter"; @@ -72,6 +73,7 @@ export function buildOpenAIChatPassthroughRequest( for (const field of CHAT_PASSTHROUGH_FIELDS) { if (rawBody[field] !== undefined) body[field] = rawBody[field]; } + if (protectGlmSummaryBudget(body)) body.reasoning_effort = "low"; const rawEfforts = modelRecordValue(provider.modelReasoningEfforts, modelId) ?? provider.reasoningEfforts; const reasoningDisabled = modelInList(provider.noReasoningModels, modelId) || rawEfforts?.length === 0; if (reasoningDisabled) { diff --git a/src/adapters/openai-chat/summary-budget.ts b/src/adapters/openai-chat/summary-budget.ts new file mode 100644 index 00000000000..82036541be6 --- /dev/null +++ b/src/adapters/openai-chat/summary-budget.ts @@ -0,0 +1,46 @@ +import { modelRecordValue } from "../../reasoning-effort"; +import type { OcxParsedRequest, OcxProviderConfig } from "../../types"; + +export function resolveMaxTokens(provider: OcxProviderConfig, parsed: OcxParsedRequest): number | undefined { + return parsed.options.maxOutputTokens + ?? modelRecordValue(provider.modelMaxOutputTokens, parsed.modelId) + ?? provider.defaultMaxOutputTokens; +} + +function textContent(value: unknown): string | undefined { + if (typeof value === "string") return value; + if (!Array.isArray(value)) return undefined; + const text: string[] = []; + for (const part of value) { + if (!part || part.type !== "text" || typeof part.text !== "string") return undefined; + text.push(part.text); + } + return text.join("\n"); +} + +/** Aside's emergency checkpoint is a standalone summary, not an ordinary short answer. + * Runs at the physical Chat destination, after all combo effort overrides. + */ +export function protectGlmSummaryBudget(body: Record): boolean { + if (typeof body.model !== "string" + || !/^(?:(?:zai|z-ai|zai-org)\/)?glm-5\.3-flash$/i.test(body.model)) return false; + if (body.tools !== undefined && (!Array.isArray(body.tools) || body.tools.length > 0)) return false; + const cap = body.max_completion_tokens ?? body.max_tokens; + if (typeof cap !== "number" || !Number.isInteger(cap) || cap < 1 || cap > 1024) return false; + const messages = body.messages; + if (!Array.isArray(messages) || messages.length !== 2) return false; + const [system, user] = messages; + if (!system || !user || !["system", "developer"].includes(system.role) || user.role !== "user" + || system.tool_calls || user.tool_calls || system.function_call || user.function_call) return false; + const instruction = textContent(system.content); + const transcript = textContent(user.content); + if (instruction === undefined || transcript === undefined) return false; + const summaryInstruction = /\bcontext[-\s]+summari[sz](?:ation|er|ing)\b/i.test(instruction); + const checkpointTranscript = /\b(?:summari[sz]e|summary|checkpoint)\b/i.test(instruction) + && /[\s\S]*<\/conversation>/i.test(transcript); + if (!summaryInstruction && !checkpointTranscript) return false; + // Update both if supplied: gateways differ on which cap takes precedence. + if (body.max_tokens !== undefined) body.max_tokens = 4096; + if (body.max_completion_tokens !== undefined) body.max_completion_tokens = 4096; + return true; +} diff --git a/tests/adapters/openai/openai-chat-glm-summary.test.ts b/tests/adapters/openai/openai-chat-glm-summary.test.ts new file mode 100644 index 00000000000..6a93df3b7d5 --- /dev/null +++ b/tests/adapters/openai/openai-chat-glm-summary.test.ts @@ -0,0 +1,65 @@ +import { describe, expect, test } from "bun:test"; +import { buildOpenAIChatPassthroughRequest, createOpenAIChatAdapter } from "../../../src/adapters/openai-chat"; +import { chatCompletionsToResponsesBody } from "../../../src/chat/inbound"; +import { concreteComboRequestBody } from "../../../src/combos/request"; +import { parseRequest } from "../../../src/responses/parser"; +import type { OcxProviderConfig } from "../../../src/types"; + +const model = "glm-5.3-flash"; +const provider: OcxProviderConfig = { + adapter: "openai-chat", baseUrl: "https://api.z.ai/api/coding/paas/v4", + reasoningEfforts: ["low", "medium", "high", "max"], +}; +const messages = [ + { role: "system", content: "You are a context-summarization assistant. Produce a checkpoint." }, + { role: "user", content: "User: implement the feature." }, +]; +function bodies(overrides: Record = {}, config = provider) { + const raw = { model, messages, max_tokens: 512, reasoning_effort: "max", ...overrides }; + const parsed = parseRequest(chatCompletionsToResponsesBody(raw)); + return [ + JSON.parse(createOpenAIChatAdapter(config).buildRequest(parsed).body), + JSON.parse(buildOpenAIChatPassthroughRequest(config, raw, String(raw.model), false).body), + ]; +} + +describe("GLM tiny standalone summary compatibility", () => { + test.each([1, 512, 819, 1024])("raises cap %i and lowers effort on both Chat paths", cap => { + for (const body of bodies({ max_tokens: cap })) { + expect(body.max_tokens).toBe(4096); + expect(body.reasoning_effort).toBe("low"); + expect(body.messages).toEqual(messages); + } + }); + test("final adapter wins after successive combo force overrides", () => { + let raw = chatCompletionsToResponsesBody({ model, messages, max_tokens: 819, reasoning_effort: "high" }); + for (const target of [{ provider: "proxy", model: "inner" }, { provider: "zai", model }]) { + raw = concreteComboRequestBody(raw, target, "max", provider.reasoningEfforts, "strict", "force"); + } + const parsed = parseRequest(raw); + parsed.modelId = model; + expect(parsed.options.reasoning).toBe("max"); + const body = JSON.parse(createOpenAIChatAdapter(provider).buildRequest(parsed).body); + expect(body.max_tokens).toBe(4096); + expect(body.reasoning_effort).toBe("low"); + expect(parsed.options.maxOutputTokens).toBe(819); + expect(parsed.options.reasoning).toBe("max"); + }); + test.each([0, -1, 1025, 4096, undefined])("preserves cap outside the mitigation: %s", cap => { + for (const body of bodies({ max_tokens: cap })) { + expect(body.max_tokens).toBe(cap); + expect(body.reasoning_effort).toBe("max"); + } + }); + test("does not change other models, ordinary prompts, tools, or ongoing conversations", () => { + for (const overrides of [ + { model: "glm-5.3" }, { model: "glm-5.3-flashx" }, { model: "gpt-5" }, + { messages: [{ role: "system", content: "Be helpful." }, { role: "user", content: "Summarize this article." }] }, + { messages: [...messages, { role: "assistant", content: "Previous checkpoint" }] }, + { tools: [{ type: "function", function: { name: "read", parameters: { type: "object", properties: {} } } }] }, + ]) for (const body of bodies(overrides)) { + expect(body.max_tokens).toBe(512); + expect(body.reasoning_effort).toBe("max"); + } + }); +}); From fb3faab2058719cc96ad2a069999e1ac238ff6d7 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:32:24 +0900 Subject: [PATCH 46/75] fix(adapters): narrow the GLM summary budget mitigation Follow-up to #5953, from the review that held it. The mitigation now applies only on Z.AI endpoints, only at an effective high or max effort, and only to the checkpoint shape: a summary instruction plus a transcript of at least 2000 characters. A summarization prompt with a short user message is left alone. Each tiny cap field (1-1024) is raised on its own to 8192, so a larger caller cap is never shrunk. Negative tests cover each boundary. --- src/adapters/openai-chat.ts | 3 +- src/adapters/openai-chat/passthrough.ts | 2 +- src/adapters/openai-chat/summary-budget.ts | 50 ++++++++++++++----- .../openai/openai-chat-glm-summary.test.ts | 39 +++++++++++++-- 4 files changed, 77 insertions(+), 17 deletions(-) diff --git a/src/adapters/openai-chat.ts b/src/adapters/openai-chat.ts index 85dbe21c569..6835be4354c 100644 --- a/src/adapters/openai-chat.ts +++ b/src/adapters/openai-chat.ts @@ -145,7 +145,8 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd body.stop = parsed.options.stopSequences; } const reasoningDisabled = modelInList(provider.noReasoningModels, parsed.modelId); - const requestedEffort = protectGlmSummaryBudget(body) ? "low" : parsed.options.reasoning; + const requestedEffort = protectGlmSummaryBudget(body, provider.baseUrl, parsed.options.reasoning) + ? "low" : parsed.options.reasoning; const reasoningEffort = mapReasoningEffort(provider, parsed.modelId, requestedEffort); const explicitReasoning = applyExplicitChatReasoningWirePolicy({ provider, diff --git a/src/adapters/openai-chat/passthrough.ts b/src/adapters/openai-chat/passthrough.ts index 9350ada953e..ed01bf13a6c 100644 --- a/src/adapters/openai-chat/passthrough.ts +++ b/src/adapters/openai-chat/passthrough.ts @@ -73,7 +73,7 @@ export function buildOpenAIChatPassthroughRequest( for (const field of CHAT_PASSTHROUGH_FIELDS) { if (rawBody[field] !== undefined) body[field] = rawBody[field]; } - if (protectGlmSummaryBudget(body)) body.reasoning_effort = "low"; + if (protectGlmSummaryBudget(body, provider.baseUrl, body.reasoning_effort)) body.reasoning_effort = "low"; const rawEfforts = modelRecordValue(provider.modelReasoningEfforts, modelId) ?? provider.reasoningEfforts; const reasoningDisabled = modelInList(provider.noReasoningModels, modelId) || rawEfforts?.length === 0; if (reasoningDisabled) { diff --git a/src/adapters/openai-chat/summary-budget.ts b/src/adapters/openai-chat/summary-budget.ts index 82036541be6..2e044a46421 100644 --- a/src/adapters/openai-chat/summary-budget.ts +++ b/src/adapters/openai-chat/summary-budget.ts @@ -18,15 +18,43 @@ function textContent(value: unknown): string | undefined { return text.join("\n"); } -/** Aside's emergency checkpoint is a standalone summary, not an ordinary short answer. - * Runs at the physical Chat destination, after all combo effort overrides. +const PROTECTED_SUMMARY_CAP = 8192; +/** A real checkpoint carries the conversation it summarizes; a short probe is not one (#5465). */ +const MIN_CHECKPOINT_TRANSCRIPT_CHARS = 2000; + +/** The mitigation is scoped to Z.AI's own endpoints; another gateway serving the same model id is not. */ +function isZaiEndpoint(baseUrl: string | undefined): boolean { + if (!baseUrl) return false; + try { + const host = new URL(baseUrl).hostname.toLowerCase(); + return host === "z.ai" || host.endsWith(".z.ai"); + } catch { + return false; + } +} + +function isTinyCap(value: unknown): value is number { + return typeof value === "number" && Number.isInteger(value) && value >= 1 && value <= 1024; +} + +/** + * Aside's emergency checkpoint is a standalone summary, not an ordinary short answer (#5465). + * Runs at the physical Chat destination, after all combo effort overrides, so `effort` is the + * effective effort. Only the exhausting tiers (`high`/`max`) on Z.AI's GLM-5.3-Flash, for the + * two-message checkpoint shape with a real `` transcript, qualify. Each tiny cap + * field is raised on its own, so a caller's larger cap is never shrunk. */ -export function protectGlmSummaryBudget(body: Record): boolean { +export function protectGlmSummaryBudget( + body: Record, + baseUrl: string | undefined, + effort: unknown, +): boolean { + if (!isZaiEndpoint(baseUrl)) return false; + if (effort !== "high" && effort !== "max") return false; if (typeof body.model !== "string" || !/^(?:(?:zai|z-ai|zai-org)\/)?glm-5\.3-flash$/i.test(body.model)) return false; if (body.tools !== undefined && (!Array.isArray(body.tools) || body.tools.length > 0)) return false; - const cap = body.max_completion_tokens ?? body.max_tokens; - if (typeof cap !== "number" || !Number.isInteger(cap) || cap < 1 || cap > 1024) return false; + if (!isTinyCap(body.max_tokens) && !isTinyCap(body.max_completion_tokens)) return false; const messages = body.messages; if (!Array.isArray(messages) || messages.length !== 2) return false; const [system, user] = messages; @@ -35,12 +63,10 @@ export function protectGlmSummaryBudget(body: Record): boolean const instruction = textContent(system.content); const transcript = textContent(user.content); if (instruction === undefined || transcript === undefined) return false; - const summaryInstruction = /\bcontext[-\s]+summari[sz](?:ation|er|ing)\b/i.test(instruction); - const checkpointTranscript = /\b(?:summari[sz]e|summary|checkpoint)\b/i.test(instruction) - && /[\s\S]*<\/conversation>/i.test(transcript); - if (!summaryInstruction && !checkpointTranscript) return false; - // Update both if supplied: gateways differ on which cap takes precedence. - if (body.max_tokens !== undefined) body.max_tokens = 4096; - if (body.max_completion_tokens !== undefined) body.max_completion_tokens = 4096; + if (!/\b(?:summari[sz](?:e|ation|er|ing)|summary|checkpoint)\b/i.test(instruction)) return false; + if (!/[\s\S]*<\/conversation>/i.test(transcript)) return false; + if (transcript.length < MIN_CHECKPOINT_TRANSCRIPT_CHARS) return false; + if (isTinyCap(body.max_tokens)) body.max_tokens = PROTECTED_SUMMARY_CAP; + if (isTinyCap(body.max_completion_tokens)) body.max_completion_tokens = PROTECTED_SUMMARY_CAP; return true; } diff --git a/tests/adapters/openai/openai-chat-glm-summary.test.ts b/tests/adapters/openai/openai-chat-glm-summary.test.ts index 6a93df3b7d5..b892d3a04f5 100644 --- a/tests/adapters/openai/openai-chat-glm-summary.test.ts +++ b/tests/adapters/openai/openai-chat-glm-summary.test.ts @@ -10,9 +10,11 @@ const provider: OcxProviderConfig = { adapter: "openai-chat", baseUrl: "https://api.z.ai/api/coding/paas/v4", reasoningEfforts: ["low", "medium", "high", "max"], }; +// Aside's emergency checkpoint: a summary instruction plus the transcript it summarizes (#5465). +const transcript = `${"User: implement the feature and keep the tests green.\n".repeat(60)}`; const messages = [ { role: "system", content: "You are a context-summarization assistant. Produce a checkpoint." }, - { role: "user", content: "User: implement the feature." }, + { role: "user", content: transcript }, ]; function bodies(overrides: Record = {}, config = provider) { const raw = { model, messages, max_tokens: 512, reasoning_effort: "max", ...overrides }; @@ -26,7 +28,7 @@ function bodies(overrides: Record = {}, config = provider) { describe("GLM tiny standalone summary compatibility", () => { test.each([1, 512, 819, 1024])("raises cap %i and lowers effort on both Chat paths", cap => { for (const body of bodies({ max_tokens: cap })) { - expect(body.max_tokens).toBe(4096); + expect(body.max_tokens).toBe(8192); expect(body.reasoning_effort).toBe("low"); expect(body.messages).toEqual(messages); } @@ -40,7 +42,7 @@ describe("GLM tiny standalone summary compatibility", () => { parsed.modelId = model; expect(parsed.options.reasoning).toBe("max"); const body = JSON.parse(createOpenAIChatAdapter(provider).buildRequest(parsed).body); - expect(body.max_tokens).toBe(4096); + expect(body.max_tokens).toBe(8192); expect(body.reasoning_effort).toBe("low"); expect(parsed.options.maxOutputTokens).toBe(819); expect(parsed.options.reasoning).toBe("max"); @@ -63,3 +65,34 @@ describe("GLM tiny standalone summary compatibility", () => { } }); }); + +describe("GLM summary mitigation stays inside its boundary (#5953 review)", () => { + const untouched = (body: Record, effort = "max") => { + expect(body.max_tokens).toBe(512); + expect(body.reasoning_effort).toBe(effort); + }; + test("another gateway serving the same model id is left alone", () => { + for (const baseUrl of ["https://example.com/v1", "https://evilz.ai/api/v4", "not a url"]) { + for (const body of bodies({}, { ...provider, baseUrl })) untouched(body); + } + }); + test("a summarization system prompt with an ordinary short user message is not a checkpoint", () => { + const probe = [messages[0], { role: "user", content: "Hello" }]; + for (const body of bodies({ messages: probe })) untouched(body); + }); + test("a checkpoint transcript under the minimum length is not rewritten", () => { + const short = [messages[0], { role: "user", content: "User: hi." }]; + for (const body of bodies({ messages: short })) untouched(body); + }); + test.each(["low", "medium"])("an effective %s effort is not overridden", effort => { + for (const body of bodies({ reasoning_effort: effort })) untouched(body, effort); + }); + test("each cap field is judged on its own", () => { + const [, mixedLarge] = bodies({ max_tokens: 512, max_completion_tokens: 2048 }); + expect(mixedLarge.max_tokens).toBe(8192); + expect(mixedLarge.max_completion_tokens).toBe(2048); + const [, mixedSmall] = bodies({ max_tokens: 4096, max_completion_tokens: 512 }); + expect(mixedSmall.max_tokens).toBe(4096); + expect(mixedSmall.max_completion_tokens).toBe(8192); + }); +}); From 25e01a8e2baf0246df1a266673d270177608589e Mon Sep 17 00:00:00 2001 From: codingbo Date: Sun, 27 Sep 2026 16:32:45 +0900 Subject: [PATCH 47/75] feat(prompt): snapshot skills catalog per session to preserve prompt cache (#6027) Carried from #6027 into merge train round 3. The layout registries were unioned with the entries that landed first. Co-authored-by: codingbo --- .../src/content/docs/guides/codex-prompt.md | 29 +++ .../docs/reference/configuration/agents.md | 9 + scripts/test-layout/layout.json | 2 +- src/config/diagnostics.ts | 15 +- src/config/schema/config-schema.ts | 2 + src/config/schema/leaf-validators.ts | 7 + src/server/responses/request-prepare.ts | 9 + src/server/responses/skills-snapshot.ts | 211 ++++++++++++++++++ src/types.ts | 2 + src/types/config.ts | 14 ++ structure/config.md | 6 +- structure/transports/responses.md | 2 +- .../config-skills-catalog-refresh.test.ts | 115 ++++++++++ tests/fixtures/test-layout-expected.json | 1 + tests/helpers/responses-core-source.ts | 1 + .../responses-skills-snapshot.test.ts | 113 ++++++++++ 16 files changed, 532 insertions(+), 6 deletions(-) create mode 100644 src/server/responses/skills-snapshot.ts create mode 100644 tests/config/config-skills-catalog-refresh.test.ts create mode 100644 tests/responses/responses-skills-snapshot.test.ts diff --git a/docs-site/src/content/docs/guides/codex-prompt.md b/docs-site/src/content/docs/guides/codex-prompt.md index df6c5a770f8..e31bf030cfd 100644 --- a/docs-site/src/content/docs/guides/codex-prompt.md +++ b/docs-site/src/content/docs/guides/codex-prompt.md @@ -169,6 +169,35 @@ repair writes a backup before it touches anything. Changes apply to newly started sessions. A session already running keeps the prompt settings it started with. +## Keeping the skills catalog stable + +The proxy defaults to `skills.catalog_refresh: "per_session"`: the first +`` catalog received for a conversation is reused on later +requests in that conversation. This keeps skill discovery and `SKILL.md` edits +from changing that part of the upstream prompt cache prefix mid-session. + +To use the catalog supplied by the client on every turn, set this in opencodex's +`$OPENCODEX_HOME/config.json` (normally `~/.opencodex/config.json`), then restart +the proxy: + +```json +{ + "skills": { + "catalog_refresh": "per_turn" + } +} +``` + +The supported values are `"per_session"` (default) and `"per_turn"`. This is a +proxy setting, separate from Codex's `skills.include_instructions` toggle. +Requests without a reliable conversation identity use the catalog supplied by +the client. Snapshots are held in memory and do not survive a proxy restart. +They expire after four hours of inactivity and may be evicted when the bounded +cache fills. An initial catalog block larger than 512 KiB is forwarded without caching. +After expiry or eviction, the next received catalog becomes the new snapshot. +The dashboard's prompt preview still reads the current files; it does not show +the snapshot retained for an ongoing conversation. + ## What this page reads, and what it does not opencodex reads one configuration file — your `config.toml`. Codex resolves its diff --git a/docs-site/src/content/docs/reference/configuration/agents.md b/docs-site/src/content/docs/reference/configuration/agents.md index ac968ee7073..20165746900 100644 --- a/docs-site/src/content/docs/reference/configuration/agents.md +++ b/docs-site/src/content/docs/reference/configuration/agents.md @@ -8,6 +8,15 @@ routes, and limits delegated work. ## Agent fields +### Skills catalog refresh + +`skills.catalog_refresh` accepts `"per_session"` (the default) or `"per_turn"` +in opencodex's `config.json`. Session mode reuses the first received skills +instructions for a conversation, protecting the prompt cache prefix from catalog +changes between turns. Turn mode forwards the client's current catalog. +See [Keeping the skills catalog stable](/guides/codex-prompt/#keeping-the-skills-catalog-stable) +for configuration and snapshot lifetime details. + ### Astra roster upgrade On the first start after upgrading, existing `subagentModels` lists receive diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 53bcb676449..cb4c2759e87 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -173,7 +173,7 @@ "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", "restart-replacement.test.ts": "server", "deepseek-artifact-tool-schema.test.ts": "providers", - "client-config-export-output-limit.test.ts": "config", + "client-config-export-output-limit.test.ts": "config", "responses-skills-snapshot.test.ts": "responses", "openai-chat-serialized-tool-call-scaling.test.ts": "adapters/openai", "openai-chat-tool-call-id-remint.test.ts": "adapters/openai", "coding-agent-json-lines-scaling.test.ts": "providers", diff --git a/src/config/diagnostics.ts b/src/config/diagnostics.ts index ca7b899e3d6..d1ab81747c4 100644 --- a/src/config/diagnostics.ts +++ b/src/config/diagnostics.ts @@ -63,6 +63,7 @@ import { runtimeRoleSchema, spendSchema, compactionRoutingSchema, + skillsConfigSchema, } from "./schema/leaf-validators"; export type ConfigDiagnostics = { @@ -594,6 +595,17 @@ export function metricsExportConfigError(value: unknown): string | null { return null; } + +function skillsConfigError(value: unknown): string | null { + const raw = rawConfigRecord(value); + if (!raw || !Object.hasOwn(raw, "skills") || raw.skills === undefined) return null; + const result = skillsConfigSchema.safeParse(raw.skills); + if (result.success) return null; + const issue = result.error.issues[0]; + const field = issue?.path.join("."); + return "schema_invalid: skills" + (field ? "." + field : "") + ": " + (issue?.message ?? "invalid configuration"); +} + export function validateConfigCandidate(value: unknown): { ok: true; config: OcxConfig } | { ok: false; error: string } { const compactionRouting = rawConfigRecord(value)?.compactionRouting; if (compactionRouting !== undefined && !compactionRoutingSchema.safeParse(compactionRouting).success) { @@ -625,7 +637,8 @@ export function validateConfigCandidate(value: unknown): { ok: true; config: Ocx ?? clientRolePairError(value) ?? loopbackListenerPortError(value) ?? managementIngressConfigError(value) - ?? metricsExportConfigError(value); + ?? metricsExportConfigError(value) + ?? skillsConfigError(value); if (boundaryError) return { ok: false, error: boundaryError }; const result = configSchema.safeParse(value); if (result.success) { diff --git a/src/config/schema/config-schema.ts b/src/config/schema/config-schema.ts index 48460d3667e..a855202a29a 100644 --- a/src/config/schema/config-schema.ts +++ b/src/config/schema/config-schema.ts @@ -16,6 +16,7 @@ import { remoteGuiConfigSchema, runtimeRoleSchema, spendSchema, + skillsConfigSchema, configuredCodexPoolAccountIds, apiKeyEntrySchema, asideProfileSyncSchema, @@ -78,6 +79,7 @@ export const configSchema = z.object({ // A malformed privacy block must never be read as "unmask": .catch(undefined) drops it and // emailMaskingEnabled then falls back to masked, which is also what an absent block means. privacy: z.object({ maskEmails: z.boolean().optional() }).strict().optional().catch(undefined), + skills: skillsConfigSchema.optional().catch(undefined), // Malformed hand edits disable this opt-in exporter. Live writes reject them in diagnostics.ts. metricsExport: z.object({ enabled: z.boolean().optional() }).strict().optional().catch(undefined), // Kept raw on purpose: `.catch(undefined)` would turn a mistyped `enabled` into "inherit", diff --git a/src/config/schema/leaf-validators.ts b/src/config/schema/leaf-validators.ts index 449fd54ce13..8de6060f5b2 100644 --- a/src/config/schema/leaf-validators.ts +++ b/src/config/schema/leaf-validators.ts @@ -1055,3 +1055,10 @@ export const spendSchema = z.object({ pool: spendScopeSchema.optional(), retentionDays: z.number().int().min(1).max(365).optional(), }).strict(); + +/** + * Runtime skills catalog configuration (#5569). + */ +export const skillsConfigSchema = z.object({ + catalog_refresh: z.enum(["per_session", "per_turn"]).optional(), +}).strict(); diff --git a/src/server/responses/request-prepare.ts b/src/server/responses/request-prepare.ts index 66369040eb3..9edb8630eb3 100644 --- a/src/server/responses/request-prepare.ts +++ b/src/server/responses/request-prepare.ts @@ -22,6 +22,7 @@ import { reasoningReplayConversationIdFromResponsesRequest, } from "../request-log-conversation"; import { resolveContextPrincipal } from "../auth-cors"; +import { resolveSkillsSnapshotScopeKey, snapshotSkillsCatalogInBody } from "./skills-snapshot"; import { isShadowSourceModel, shadowSourceModelPrefix, @@ -314,6 +315,14 @@ export async function prepareResponsesRequest( ); } + const skillsSnapshotScopeKey = resolveSkillsSnapshotScopeKey({ + req, + config, + admission: options.admission, + promptCacheKeyIsSharedCohort: options.promptCacheKeyIsSharedCohort, + }); + snapshotSkillsCatalogInBody(body, skillsSnapshotScopeKey, config); + let parsed: OcxParsedRequest; let toolBridgeMaps: ReturnType; try { diff --git a/src/server/responses/skills-snapshot.ts b/src/server/responses/skills-snapshot.ts new file mode 100644 index 00000000000..b6c7b057405 --- /dev/null +++ b/src/server/responses/skills-snapshot.ts @@ -0,0 +1,211 @@ +/** + * runtime skills catalog session snapshotting (#5569). + * + * Preserves the Anthropic/LLM prompt cache prefix across turns by freezing + * incoming for the duration of a trustworthy session. + * Gated by config `skills.catalog_refresh`: "per_session" (default) or "per_turn". + * + * Lifecycle & Bounds: + * - 4 hours idle TTL (sliding on access) + * - 1,000 maximum tracked sessions (LRU eviction) + * - 512 KiB maximum per snapshotted skills block + * - 8 MiB global retained byte bound across all sessions + */ +import type { OcxConfig, SkillsCatalogRefresh } from "../../types/config"; +import { resolveContextPrincipal, type DataPlaneAdmission } from "../auth-cors"; +import { + reasoningReplayConversationIdFromResponsesRequest, + sessionIdHeaderFromRequest, +} from "../request-log-conversation"; + +const SKILLS_BLOCK_GLOBAL_REGEX = /([\s\S]*?)<\/skills_instructions>/g; + +/** Maximum distinct sessions tracked in the memory LRU. */ +export const MAX_SNAPSHOT_SESSIONS = 1000; +/** Slide expiry after 4 hours of inactivity. */ +export const SNAPSHOT_TTL_MS = 4 * 60 * 60 * 1000; +/** Bounded byte ceiling per snapshotted skills block (512 KiB). */ +export const MAX_SKILLS_BLOCK_BYTES = 512 * 1024; +/** Global retained byte bound across all tracked sessions (8 MiB). */ +export const MAX_TOTAL_RETAINED_BYTES = 8 * 1024 * 1024; + +interface SnapshotEntry { + skillsBlock: string; // The full ... block + byteLength: number; + lastAccessed: number; +} + +const snapshotCache = new Map(); +let totalRetainedBytes = 0; + +function evictOldestEntry(): boolean { + const oldest = snapshotCache.entries().next().value; + if (!oldest) return false; + const [key, entry] = oldest; + totalRetainedBytes -= entry.byteLength; + snapshotCache.delete(key); + return true; +} + +export function resolveSkillsCatalogRefresh(config: OcxConfig | undefined): SkillsCatalogRefresh { + const configured = config?.skills?.catalog_refresh; + if (configured === "per_turn") return "per_turn"; + return "per_session"; +} + +export interface ResolveSkillsSessionScopeInput { + req: Request; + config: OcxConfig; + admission?: DataPlaneAdmission; + cursorConversationId?: string; + promptCacheKeyIsSharedCohort?: boolean; +} + +/** + * Resolves a trustworthy cache key for skills catalog snapshotting. + * Returns null if no specific, reliable thread/session identity is available, + * or if the identity comes from a shared cohort fallback. + */ +export function resolveSkillsSnapshotScopeKey(input: ResolveSkillsSessionScopeInput): string | null { + if (input.promptCacheKeyIsSharedCohort === true) { + return null; + } + + const parentThread = input.req.headers.get("x-codex-parent-thread-id")?.trim() || undefined; + const ownThreadId = input.req.headers.get("thread-id")?.trim() || undefined; + const cursorId = input.cursorConversationId?.trim() || undefined; + + // When a parent thread is present, subagents/children may share a root session-id header. + // To prevent cross-thread/sibling collapse or parent-level caching, require an explicit + // own thread-id (or cursor id). If parentThread is present without an own child thread, + // bypass snapshotting completely. + if (parentThread) { + const childId = ownThreadId ?? cursorId; + if (!childId || childId === parentThread) { + return null; + } + const qualifiedId = `${parentThread}\u0000${childId}`; + const principal = resolveContextPrincipal(input.req, input.config, input.admission) ?? null; + return JSON.stringify(["skills_catalog_snapshot_v1", principal, qualifiedId]); + } + + // Standalone conversation (no parent thread) + const standaloneId = reasoningReplayConversationIdFromResponsesRequest({ + threadIdHeader: ownThreadId, + cursorConversationId: cursorId, + sessionIdHeader: sessionIdHeaderFromRequest(input.req.headers), + }); + if (!standaloneId) { + return null; + } + const principal = resolveContextPrincipal(input.req, input.config, input.admission) ?? null; + return JSON.stringify(["skills_catalog_snapshot_v1", principal, standaloneId]); +} + +function snapshotOrReplaceInText( + text: string, + scopeKey: string, + now: number, +): string { + if (!text.includes("")) return text; + + return text.replace(SKILLS_BLOCK_GLOBAL_REGEX, (match) => { + const existing = snapshotCache.get(scopeKey); + if (existing) { + // Check TTL on cache hits + if (now - existing.lastAccessed > SNAPSHOT_TTL_MS) { + totalRetainedBytes -= existing.byteLength; + snapshotCache.delete(scopeKey); + } else { + existing.lastAccessed = now; + // Refresh Map order for true LRU behavior + snapshotCache.delete(scopeKey); + snapshotCache.set(scopeKey, existing); + return existing.skillsBlock; + } + } + + // First turn or expired: snapshot incoming block if bounded + const incomingBlock = match; + const blockBytes = Buffer.byteLength(incomingBlock, "utf8"); + if (blockBytes <= MAX_SKILLS_BLOCK_BYTES && blockBytes <= MAX_TOTAL_RETAINED_BYTES) { + // Evict oldest entries until under count ceiling AND under global byte ceiling + while ( + (snapshotCache.size >= MAX_SNAPSHOT_SESSIONS || totalRetainedBytes + blockBytes > MAX_TOTAL_RETAINED_BYTES) + && snapshotCache.size > 0 + ) { + if (!evictOldestEntry()) break; + } + + if (totalRetainedBytes + blockBytes <= MAX_TOTAL_RETAINED_BYTES) { + snapshotCache.set(scopeKey, { + skillsBlock: incomingBlock, + byteLength: blockBytes, + lastAccessed: now, + }); + totalRetainedBytes += blockBytes; + } + } + return match; + }); +} + +/** + * Transforms incoming developer/system prompt contents to reuse the session's + * snapshotted , preserving prefix cache across turns. + * User and assistant messages, as well as tool calls, are never modified. + */ +export function snapshotSkillsCatalogInBody( + body: unknown, + scopeKey: string | null, + config: OcxConfig, + now: number = Date.now(), +): void { + if (!scopeKey) return; + if (resolveSkillsCatalogRefresh(config) === "per_turn") return; + if (!body || typeof body !== "object" || Array.isArray(body)) return; + + const b = body as Record; + + // 1. Check top-level instructions field + if (typeof b.instructions === "string" && b.instructions.includes("")) { + b.instructions = snapshotOrReplaceInText(b.instructions, scopeKey, now); + } + + // 2. Check input array for developer/system messages only + if (Array.isArray(b.input)) { + for (const item of b.input) { + if (!item || typeof item !== "object") continue; + const it = item as Record; + // Restrict message item type: must be undefined or "message", so role-like tool objects are untouched + if (it.type !== undefined && it.type !== "message") continue; + const role = it.role; + // Only developer and system content is inspected/transformed + if (role !== "developer" && role !== "system") continue; + + const content = it.content; + if (typeof content === "string") { + if (content.includes("")) { + it.content = snapshotOrReplaceInText(content, scopeKey, now); + } + } else if (Array.isArray(content)) { + for (const part of content) { + if (!part || typeof part !== "object") continue; + const p = part as Record; + // Restrict text parts to known text / input_text + if (p.type !== "text" && p.type !== "input_text") continue; + if (typeof p.text === "string" && p.text.includes("")) { + p.text = snapshotOrReplaceInText(p.text, scopeKey, now); + } + } + } + } + } +} + +/** Test helpers */ +export function resetSkillsSnapshotCacheForTests(): void { + snapshotCache.clear(); + totalRetainedBytes = 0; +} + diff --git a/src/types.ts b/src/types.ts index ff961d813e4..5281666884d 100644 --- a/src/types.ts +++ b/src/types.ts @@ -77,6 +77,8 @@ export type { OcxConnectedClientId, OcxClientConnectionConfig, OcxConfig, + SkillsCatalogRefresh, + OcxSkillsConfig, OcxAccountPoolRotationStrategy, OcxAccountPoolQuotaWindow, OcxComboCooldownWaitPolicy, diff --git a/src/types/config.ts b/src/types/config.ts index 6e7fa9a48ac..940aa2b06fd 100644 --- a/src/types/config.ts +++ b/src/types/config.ts @@ -362,6 +362,18 @@ export interface OcxConfigRebaseProvenance { deletedTopLevelKeys: string[]; } + +export type SkillsCatalogRefresh = "per_session" | "per_turn"; + +export interface OcxSkillsConfig { + /** + * Refresh policy for the runtime skills catalog (#5569). + * `per_session` (default): snapshots incoming `` on the first turn of a trustworthy session and reuses it across turns to preserve the Anthropic prompt cache. + * `per_turn`: re-derives/passes through incoming skills instructions every turn (previous behavior). + */ + catalog_refresh?: SkillsCatalogRefresh; +} + export type OcxRuntimeRole = "standalone" | "hub" | "client"; export interface OcxHubConfig { @@ -487,6 +499,8 @@ export interface OcxConfig { client?: OcxClientConnectionConfig; /** Operator-facing redaction policy for management and CLI projections. */ privacy?: OcxPrivacyConfig; + /** Runtime skills catalog session snapshotting settings (#5569). */ + skills?: OcxSkillsConfig; /** Opt-in process-local aggregate request metrics on the authenticated management plane. */ metricsExport?: { enabled?: boolean }; /** Opt in to one identical-turn retry when a Responses completion has no text or tool call. */ diff --git a/structure/config.md b/structure/config.md index 3ce49bf31f3..5088fbf4ffe 100644 --- a/structure/config.md +++ b/structure/config.md @@ -30,10 +30,10 @@ the [source-owned credential contract](codex-home.md#orca-source-owned-account-i `src/config/schema/compaction-recovery.ts` strictly validates opt-in `compactionRecovery`; invalid disk values disable it with a warning, while candidate writes reject them. The [failure-only contract](transports/responses-failover.md) leaves provider identity, accounts and client compaction unchanged. +`skills.catalog_refresh` in the proxy JSON configuration accepts `per_session` (the runtime default when absent) or `per_turn`. The former retains received skills instructions for a conversation; the latter passes through the current catalog. This is separate from Codex's `skills.include_instructions` TOML switch and does not change the live dashboard probe. See the [Responses snapshot contract](transports/responses.md#responses-httpsse). + Google providers may persist `googleToolSchemaPolicy` as `compatible` or `reject-lossy`. -`ocx provider add --google-tool-schema-policy` is one authoring path and is accepted only when the -effective adapter is `google`. Omission remains absent in `config.json`; the adapter resolves it to -`compatible` in memory. +`ocx provider add --google-tool-schema-policy` is one authoring path and is accepted only when the effective adapter is `google`. Omission remains absent in `config.json`; the adapter resolves it to `compatible` in memory. ### OpenCodex home and live process state diff --git a/structure/transports/responses.md b/structure/transports/responses.md index 8320d296bac..2ecf589bb77 100644 --- a/structure/transports/responses.md +++ b/structure/transports/responses.md @@ -11,7 +11,7 @@ Plaintext collaboration restoration treats a null namespace as absent, rejects n When a successful streamed native response has a missing or unrecognized non-JSON content type, the plaintext V2 path confirms a bounded Responses SSE prefix, under the server's `stallTimeoutSec` probe budget, before applying that restoration; an `application/json` body takes the bounded JSON path instead, and an unknown, stalled, or unreadable body retains the fail-closed response. ## Responses HTTP/SSE - +Responses request preparation stabilizes incoming `` under `skills.catalog_refresh`: `per_session` (default) reuses the first received catalog for a conversation; `per_turn` leaves the supplied catalog unchanged. Other instruction sections and user/tool content remain untouched. Requests without a reliable conversation identity bypass snapshots; shared prompt-cache cohorts are not conversation identities. Snapshots are process-local, expire after four idle hours, and use bounded LRU retention; oversized blocks bypass caching. The dashboard's `src/codex/prompt-layers.ts` and `src/codex/prompt-text-probe.ts` continue observing current files for previews and do not own session snapshots. `/v1/responses` is the main Codex-facing endpoint. The server parses Responses input, routes to a provider, lets the selected adapter speak the upstream protocol, then bridges adapter events back to Responses-compatible streaming output. For an opted-in key-auth provider, a hosted-search continuation stays bound to the API-key selection that served the first leg; the contract is the [hosted-search continuation binding](../providers-and-adapters.md#hosted-search-continuation-binding). diff --git a/tests/config/config-skills-catalog-refresh.test.ts b/tests/config/config-skills-catalog-refresh.test.ts new file mode 100644 index 00000000000..7ddac6ce389 --- /dev/null +++ b/tests/config/config-skills-catalog-refresh.test.ts @@ -0,0 +1,115 @@ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { mkdtempSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { + getConfigPath, + getDefaultConfig, + loadConfig, + saveConfig, + validateConfigCandidate, +} from "../../src/config"; +import { resolveSkillsCatalogRefresh } from "../../src/server/responses/skills-snapshot"; +import { removeTreeWithRetry } from "../helpers/remove-tree"; + +let home = ""; +let previousHome: string | undefined; + +beforeEach(() => { + previousHome = process.env.OPENCODEX_HOME; + home = mkdtempSync(join(tmpdir(), "ocx-skills-config-")); + process.env.OPENCODEX_HOME = home; +}); + +afterEach(() => { + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + removeTreeWithRetry(home); +}); + +function candidate(skills: unknown) { + return { + ...getDefaultConfig(), + defaultProvider: "xai", + providers: { + xai: { + adapter: "openai-responses", + baseUrl: "https://api.x.ai/v1", + }, + }, + skills, + }; +} + +test("validateConfigCandidate accepts valid skills configuration", () => { + const c1 = validateConfigCandidate(candidate({ catalog_refresh: "per_session" })); + expect(c1.ok).toBe(true); + if (c1.ok) { + expect(c1.config.skills?.catalog_refresh).toBe("per_session"); + } + + const c2 = validateConfigCandidate(candidate({ catalog_refresh: "per_turn" })); + expect(c2.ok).toBe(true); + if (c2.ok) { + expect(c2.config.skills?.catalog_refresh).toBe("per_turn"); + } + + const c3 = validateConfigCandidate(candidate({})); + expect(c3.ok).toBe(true); + + const c4 = validateConfigCandidate(candidate(undefined)); + expect(c4.ok).toBe(true); +}); + +test("validateConfigCandidate explicitly rejects invalid skills configuration", () => { + const badValue = validateConfigCandidate(candidate({ catalog_refresh: "invalid_refresh" })); + expect(badValue.ok).toBe(false); + if (!badValue.ok) { + expect(badValue.error).toContain("schema_invalid: skills.catalog_refresh"); + } + + const extraProp = validateConfigCandidate(candidate({ catalog_refresh: "per_session", extra: 123 })); + expect(extraProp.ok).toBe(false); + if (!extraProp.ok) { + expect(extraProp.error).toContain("schema_invalid: skills"); + } + + const nonObject = validateConfigCandidate(candidate("per_session")); + expect(nonObject.ok).toBe(false); + if (!nonObject.ok) { + expect(nonObject.error).toContain("schema_invalid: skills"); + } +}); + +test("resolveSkillsCatalogRefresh defaults to per_session", () => { + expect(resolveSkillsCatalogRefresh(undefined)).toBe("per_session"); + expect(resolveSkillsCatalogRefresh({} as any)).toBe("per_session"); + expect(resolveSkillsCatalogRefresh({ skills: {} } as any)).toBe("per_session"); + expect(resolveSkillsCatalogRefresh({ skills: { catalog_refresh: "per_session" } } as any)).toBe("per_session"); + expect(resolveSkillsCatalogRefresh({ skills: { catalog_refresh: "per_turn" } } as any)).toBe("per_turn"); +}); + +test("skills configuration persists to disk and loads correctly", () => { + const cfg = { + ...getDefaultConfig(), + skills: { catalog_refresh: "per_turn" as const }, + }; + saveConfig(cfg); + + const loaded = loadConfig(); + expect(loaded.skills?.catalog_refresh).toBe("per_turn"); +}); + +test("malformed hand-edited skills in config file degrades gracefully on load", () => { + const configPath = getConfigPath(); + const raw = JSON.stringify({ + ...getDefaultConfig(), + skills: { catalog_refresh: "bad_value" }, + }); + writeFileSync(configPath, raw, "utf8"); + + const loaded = loadConfig(); + expect(loaded.skills).toBeUndefined(); + expect(resolveSkillsCatalogRefresh(loaded)).toBe("per_session"); +}); + diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 2924eaf5ca9..89dfe34703d 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -19,6 +19,7 @@ "deepseek-artifact-tool-schema.test.ts": "providers", "client-config-export-output-limit.test.ts": "config", "codex-account-clear-paused.test.ts": "codex-integration", + "responses-skills-snapshot.test.ts": "responses", "openai-chat-serialized-tool-call-scaling.test.ts": "adapters/openai", "openai-chat-tool-call-id-remint.test.ts": "adapters/openai", "coding-agent-json-lines-scaling.test.ts": "providers", diff --git a/tests/helpers/responses-core-source.ts b/tests/helpers/responses-core-source.ts index 7460b64ada9..f803be42c3e 100644 --- a/tests/helpers/responses-core-source.ts +++ b/tests/helpers/responses-core-source.ts @@ -30,6 +30,7 @@ export const RESPONSES_CORE_MODULES = [ "core-combo.ts", "core-combo-native.ts", "request-prepare.ts", + "skills-snapshot.ts", "shadow-target-availability.ts", "compaction-routing.ts", "compaction-recovery.ts", diff --git a/tests/responses/responses-skills-snapshot.test.ts b/tests/responses/responses-skills-snapshot.test.ts new file mode 100644 index 00000000000..b72b51efe2e --- /dev/null +++ b/tests/responses/responses-skills-snapshot.test.ts @@ -0,0 +1,113 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { getDefaultConfig } from "../../src/config/proxy-env"; +import { handleResponses } from "../../src/server/responses/core"; +import { resetSkillsSnapshotCacheForTests } from "../../src/server/responses/skills-snapshot"; +import type { OcxConfig } from "../../src/types"; +import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; + +const originalFetch = globalThis.fetch; +let releaseSpendHome: (() => void) | undefined; +const captured: string[] = []; + +beforeEach(() => { + releaseSpendHome = acquireOwnedSpendHome(); + resetSkillsSnapshotCacheForTests(); + captured.length = 0; +}); +afterEach(() => { + releaseSpendHome?.(); + releaseSpendHome = undefined; + globalThis.fetch = originalFetch; + resetSkillsSnapshotCacheForTests(); +}); + +function fixture(adapter: "openai-responses" | "anthropic"): OcxConfig { + globalThis.fetch = (async (_input, init) => { + captured.push(String(init?.body)); + return Response.json(adapter === "anthropic" ? { + id: "msg_skills", type: "message", role: "assistant", model: "fixture-model", + content: [{ type: "text", text: "done" }], stop_reason: "end_turn", + usage: { input_tokens: 10, output_tokens: 1 }, + } : { + id: "resp_skills", status: "completed", + output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "done" }] }], + usage: { input_tokens: 10, output_tokens: 1, total_tokens: 11 }, + }); + }) as typeof fetch; + return { + ...getDefaultConfig(), + defaultProvider: "fixture", + providers: { + fixture: { adapter, baseUrl: "https://fixture.test/v1", authMode: "key", apiKey: "fixture-key" }, + }, + }; +} + +async function send(config: OcxConfig, catalog: string, thread?: string, surrounding = "outside", headers: Record = {}) { + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", + headers: { + "content-type": "application/json", + ...headers, + ...(thread ? { "thread-id": thread, "x-codex-parent-thread-id": "shared-parent" } : {}), + }, + body: JSON.stringify({ + model: "fixture/fixture-model", stream: false, + input: [ + { role: "developer", content: [{ type: "input_text", text: `${surrounding}${catalog}` }] }, + { role: "user", content: "Please help with user-example" }, + ], + }), + }), config, { model: "", provider: "" }); + const text = await response.text(); + expect({ status: response.status, ...(response.status !== 200 ? { text } : {}) }).toEqual({ status: 200 }); + return captured.at(-1)!; +} + +describe("skills catalog snapshots on the Responses request path", () => { + for (const adapter of ["openai-responses", "anthropic"] as const) { + test(`${adapter}: keeps the catalog stable while preserving surrounding instructions and sibling isolation`, async () => { + const config = fixture(adapter); + await send(config, "first-catalog", "child-a"); + const second = await send(config, "edited-catalog", "child-a", "updated-outside"); + expect(second).toContain("first-catalog"); + expect(second).not.toContain("edited-catalog"); + expect(second).toContain("updated-outside"); + expect(second).toContain("user-example"); + const sibling = await send(config, "sibling-catalog", "child-b"); + expect(sibling).toContain("sibling-catalog"); + expect(sibling).not.toContain("first-catalog"); + }); + } + + test("per_turn forwards changed catalogs", async () => { + const config = fixture("anthropic"); + config.skills = { catalog_refresh: "per_turn" }; + await send(config, "first-catalog", "turn-mode"); + expect(await send(config, "edited-catalog", "turn-mode")).toContain("edited-catalog"); + }); + + test("session header aliases reuse the same snapshot", async () => { + const config = fixture("anthropic"); + await send(config, "session-catalog", undefined, "outside", { session_id: "session-a" }); + const next = await send(config, "edited-catalog", undefined, "outside", { "session-id": "session-a" }); + expect(next).toContain("session-catalog"); + expect(next).not.toContain("edited-catalog"); + expect(await send(config, "new-session-catalog", undefined, "outside", { session_id: "session-b" })) + .toContain("new-session-catalog"); + }); + + test("requests without a conversation identity never share catalogs", async () => { + const config = fixture("anthropic"); + await send(config, "first-catalog"); + expect(await send(config, "edited-catalog")).toContain("edited-catalog"); + }); + + test("a parent-only routing identity cannot share sibling catalogs", async () => { + const config = fixture("anthropic"); + const headers = { "x-codex-parent-thread-id": "parent-without-child" }; + await send(config, "first-child-catalog", undefined, "outside", headers); + expect(await send(config, "second-child-catalog", undefined, "outside", headers)) + .toContain("second-child-catalog"); + }); +}); From 01b7e24d2db202cd96dfe7cc150947b62bdc4edc Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:35:16 +0900 Subject: [PATCH 48/75] fix(prompt): snapshot one catalog, only for admitted requests Follow-up to #6027, answering the three blockers in its review. A body with more than one block across its instructions and developer/system content now passes through untouched instead of having every block rewritten to one catalog, which also bounds the substitution to one block. A known snapshot is still substituted before parsing, but a new catalog is stored only when request preparation reaches its success return, so a request rejected by parsing or admission pins nothing. Without a named principal, snapshots are shared by conversation id only on a server that requires no data-plane auth. One regression test per blocker; all three fail on the PR head. --- .../src/content/docs/guides/codex-prompt.md | 2 + src/server/responses/request-prepare.ts | 4 +- src/server/responses/skills-snapshot.ts | 202 ++++++++++-------- structure/transports/responses.md | 2 +- .../responses-skills-snapshot.test.ts | 62 +++++- 5 files changed, 182 insertions(+), 90 deletions(-) diff --git a/docs-site/src/content/docs/guides/codex-prompt.md b/docs-site/src/content/docs/guides/codex-prompt.md index e31bf030cfd..9a4e776cc91 100644 --- a/docs-site/src/content/docs/guides/codex-prompt.md +++ b/docs-site/src/content/docs/guides/codex-prompt.md @@ -175,6 +175,8 @@ The proxy defaults to `skills.catalog_refresh: "per_session"`: the first `` catalog received for a conversation is reused on later requests in that conversation. This keeps skill discovery and `SKILL.md` edits from changing that part of the upstream prompt cache prefix mid-session. +A request that carries more than one `` block is passed +through unchanged, and a request the proxy rejects does not set the catalog. To use the catalog supplied by the client on every turn, set this in opencodex's `$OPENCODEX_HOME/config.json` (normally `~/.opencodex/config.json`), then restart diff --git a/src/server/responses/request-prepare.ts b/src/server/responses/request-prepare.ts index 9edb8630eb3..2ac1679a72f 100644 --- a/src/server/responses/request-prepare.ts +++ b/src/server/responses/request-prepare.ts @@ -321,7 +321,8 @@ export async function prepareResponsesRequest( admission: options.admission, promptCacheKeyIsSharedCohort: options.promptCacheKeyIsSharedCohort, }); - snapshotSkillsCatalogInBody(body, skillsSnapshotScopeKey, config); + // Substitutes a known snapshot now; a new catalog is only stored once the request is prepared. + const commitSkillsSnapshot = snapshotSkillsCatalogInBody(body, skillsSnapshotScopeKey, config); let parsed: OcxParsedRequest; let toolBridgeMaps: ReturnType; @@ -1272,6 +1273,7 @@ export async function prepareResponsesRequest( ? admissionState.authCtx.accountId : config.activeCodexAccountId ?? null; + commitSkillsSnapshot?.(); return { inboundWire, translatorBudget, diff --git a/src/server/responses/skills-snapshot.ts b/src/server/responses/skills-snapshot.ts index b6c7b057405..0a310ae4a7a 100644 --- a/src/server/responses/skills-snapshot.ts +++ b/src/server/responses/skills-snapshot.ts @@ -12,7 +12,7 @@ * - 8 MiB global retained byte bound across all sessions */ import type { OcxConfig, SkillsCatalogRefresh } from "../../types/config"; -import { resolveContextPrincipal, type DataPlaneAdmission } from "../auth-cors"; +import { isApiAuthRequired, resolveContextPrincipal, type DataPlaneAdmission } from "../auth-cors"; import { reasoningReplayConversationIdFromResponsesRequest, sessionIdHeaderFromRequest, @@ -61,6 +61,20 @@ export interface ResolveSkillsSessionScopeInput { promptCacheKeyIsSharedCohort?: boolean; } +/** + * The principal a snapshot is scoped to. Without a named principal, only a server that requires + * no data-plane auth may share by conversation id (loopback admission, or an internal caller that + * passed none on such a server): its callers already share one trust domain, which on a no-auth + * server bound beyond loopback includes remote callers. Any other anonymous request gets no + * snapshot. + */ +function snapshotPrincipal(input: ResolveSkillsSessionScopeInput): string | null | undefined { + const principal = resolveContextPrincipal(input.req, input.config, input.admission); + if (principal) return principal; + if (input.admission) return input.admission.kind === "loopback" ? null : undefined; + return isApiAuthRequired(input.config) ? undefined : null; +} + /** * Resolves a trustworthy cache key for skills catalog snapshotting. * Returns null if no specific, reliable thread/session identity is available, @@ -85,7 +99,8 @@ export function resolveSkillsSnapshotScopeKey(input: ResolveSkillsSessionScopeIn return null; } const qualifiedId = `${parentThread}\u0000${childId}`; - const principal = resolveContextPrincipal(input.req, input.config, input.admission) ?? null; + const principal = snapshotPrincipal(input); + if (principal === undefined) return null; return JSON.stringify(["skills_catalog_snapshot_v1", principal, qualifiedId]); } @@ -98,109 +113,122 @@ export function resolveSkillsSnapshotScopeKey(input: ResolveSkillsSessionScopeIn if (!standaloneId) { return null; } - const principal = resolveContextPrincipal(input.req, input.config, input.admission) ?? null; + const principal = snapshotPrincipal(input); + if (principal === undefined) return null; return JSON.stringify(["skills_catalog_snapshot_v1", principal, standaloneId]); } -function snapshotOrReplaceInText( - text: string, - scopeKey: string, - now: number, -): string { - if (!text.includes("")) return text; - - return text.replace(SKILLS_BLOCK_GLOBAL_REGEX, (match) => { - const existing = snapshotCache.get(scopeKey); - if (existing) { - // Check TTL on cache hits - if (now - existing.lastAccessed > SNAPSHOT_TTL_MS) { - totalRetainedBytes -= existing.byteLength; - snapshotCache.delete(scopeKey); - } else { - existing.lastAccessed = now; - // Refresh Map order for true LRU behavior - snapshotCache.delete(scopeKey); - snapshotCache.set(scopeKey, existing); - return existing.skillsBlock; +/** One text slot that may carry a catalog: `instructions` or a developer/system text part. */ +interface CatalogSlot { + text: string; + write(next: string): void; +} + +/** Every developer/system text slot, walked the same way the replacement writes. */ +function catalogSlots(body: Record): CatalogSlot[] { + const slots: CatalogSlot[] = []; + if (typeof body.instructions === "string") { + slots.push({ text: body.instructions, write: next => { body.instructions = next; } }); + } + if (!Array.isArray(body.input)) return slots; + for (const item of body.input) { + if (!item || typeof item !== "object") continue; + const it = item as Record; + // Restrict message item type: must be undefined or "message", so role-like tool objects are untouched + if (it.type !== undefined && it.type !== "message") continue; + // Only developer and system content is inspected/transformed + if (it.role !== "developer" && it.role !== "system") continue; + const content = it.content; + if (typeof content === "string") { + slots.push({ text: content, write: next => { it.content = next; } }); + } else if (Array.isArray(content)) { + for (const part of content) { + if (!part || typeof part !== "object") continue; + const p = part as Record; + // Restrict text parts to known text / input_text + if (p.type !== "text" && p.type !== "input_text") continue; + if (typeof p.text === "string") slots.push({ text: p.text, write: next => { p.text = next; } }); } } + } + return slots; +} - // First turn or expired: snapshot incoming block if bounded - const incomingBlock = match; - const blockBytes = Buffer.byteLength(incomingBlock, "utf8"); - if (blockBytes <= MAX_SKILLS_BLOCK_BYTES && blockBytes <= MAX_TOTAL_RETAINED_BYTES) { - // Evict oldest entries until under count ceiling AND under global byte ceiling - while ( - (snapshotCache.size >= MAX_SNAPSHOT_SESSIONS || totalRetainedBytes + blockBytes > MAX_TOTAL_RETAINED_BYTES) - && snapshotCache.size > 0 - ) { - if (!evictOldestEntry()) break; - } +function liveSnapshot(scopeKey: string, now: number): SnapshotEntry | undefined { + const existing = snapshotCache.get(scopeKey); + if (!existing) return undefined; + if (now - existing.lastAccessed > SNAPSHOT_TTL_MS) { + totalRetainedBytes -= existing.byteLength; + snapshotCache.delete(scopeKey); + return undefined; + } + existing.lastAccessed = now; + // Refresh Map order for true LRU behavior + snapshotCache.delete(scopeKey); + snapshotCache.set(scopeKey, existing); + return existing; +} - if (totalRetainedBytes + blockBytes <= MAX_TOTAL_RETAINED_BYTES) { - snapshotCache.set(scopeKey, { - skillsBlock: incomingBlock, - byteLength: blockBytes, - lastAccessed: now, - }); - totalRetainedBytes += blockBytes; - } - } - return match; - }); +function storeSnapshot(scopeKey: string, skillsBlock: string, now: number): void { + const blockBytes = Buffer.byteLength(skillsBlock, "utf8"); + if (blockBytes > MAX_SKILLS_BLOCK_BYTES || blockBytes > MAX_TOTAL_RETAINED_BYTES) return; + // A concurrent request of the same conversation may have stored first; the first catalog wins. + if (snapshotCache.has(scopeKey)) return; + // Evict oldest entries until under count ceiling AND under global byte ceiling + while ( + (snapshotCache.size >= MAX_SNAPSHOT_SESSIONS || totalRetainedBytes + blockBytes > MAX_TOTAL_RETAINED_BYTES) + && snapshotCache.size > 0 + ) { + if (!evictOldestEntry()) break; + } + if (totalRetainedBytes + blockBytes > MAX_TOTAL_RETAINED_BYTES) return; + snapshotCache.set(scopeKey, { skillsBlock, byteLength: blockBytes, lastAccessed: now }); + totalRetainedBytes += blockBytes; } /** - * Transforms incoming developer/system prompt contents to reuse the session's - * snapshotted , preserving prefix cache across turns. - * User and assistant messages, as well as tool calls, are never modified. + * Reuses the session's snapshotted in developer/system content, preserving + * the prompt cache prefix across turns. User and assistant messages and tool calls are never + * modified. + * + * Only a body with exactly one catalog block across all of those slots takes part: with two or + * more there is no way to tell which one the snapshot stands for, so the body passes through + * untouched rather than rewriting every block to one catalog. + * + * A known snapshot is substituted at once, so parsing and the verbatim passthrough body both see + * it. A new catalog is only stored through the returned commit, which the caller runs once the + * request has passed parsing and admission, so a rejected first request pins nothing. */ export function snapshotSkillsCatalogInBody( body: unknown, scopeKey: string | null, config: OcxConfig, now: number = Date.now(), -): void { - if (!scopeKey) return; - if (resolveSkillsCatalogRefresh(config) === "per_turn") return; - if (!body || typeof body !== "object" || Array.isArray(body)) return; - - const b = body as Record; - - // 1. Check top-level instructions field - if (typeof b.instructions === "string" && b.instructions.includes("")) { - b.instructions = snapshotOrReplaceInText(b.instructions, scopeKey, now); - } - - // 2. Check input array for developer/system messages only - if (Array.isArray(b.input)) { - for (const item of b.input) { - if (!item || typeof item !== "object") continue; - const it = item as Record; - // Restrict message item type: must be undefined or "message", so role-like tool objects are untouched - if (it.type !== undefined && it.type !== "message") continue; - const role = it.role; - // Only developer and system content is inspected/transformed - if (role !== "developer" && role !== "system") continue; - - const content = it.content; - if (typeof content === "string") { - if (content.includes("")) { - it.content = snapshotOrReplaceInText(content, scopeKey, now); - } - } else if (Array.isArray(content)) { - for (const part of content) { - if (!part || typeof part !== "object") continue; - const p = part as Record; - // Restrict text parts to known text / input_text - if (p.type !== "text" && p.type !== "input_text") continue; - if (typeof p.text === "string" && p.text.includes("")) { - p.text = snapshotOrReplaceInText(p.text, scopeKey, now); - } - } - } +): (() => void) | undefined { + if (!scopeKey) return undefined; + if (resolveSkillsCatalogRefresh(config) === "per_turn") return undefined; + if (!body || typeof body !== "object" || Array.isArray(body)) return undefined; + + let found: { slot: CatalogSlot; block: string } | undefined; + let blocks = 0; + for (const slot of catalogSlots(body as Record)) { + if (!slot.text.includes("")) continue; + for (const match of slot.text.matchAll(SKILLS_BLOCK_GLOBAL_REGEX)) { + blocks++; + found ??= { slot, block: match[0] }; } } + if (blocks !== 1 || !found) return undefined; + + const existing = liveSnapshot(scopeKey, now); + if (existing) { + const { slot, block } = found; + const at = slot.text.indexOf(block); + slot.write(slot.text.slice(0, at) + existing.skillsBlock + slot.text.slice(at + block.length)); + return undefined; + } + const incoming = found.block; + return () => storeSnapshot(scopeKey, incoming, now); } /** Test helpers */ diff --git a/structure/transports/responses.md b/structure/transports/responses.md index 2ecf589bb77..aa15f709724 100644 --- a/structure/transports/responses.md +++ b/structure/transports/responses.md @@ -11,7 +11,7 @@ Plaintext collaboration restoration treats a null namespace as absent, rejects n When a successful streamed native response has a missing or unrecognized non-JSON content type, the plaintext V2 path confirms a bounded Responses SSE prefix, under the server's `stallTimeoutSec` probe budget, before applying that restoration; an `application/json` body takes the bounded JSON path instead, and an unknown, stalled, or unreadable body retains the fail-closed response. ## Responses HTTP/SSE -Responses request preparation stabilizes incoming `` under `skills.catalog_refresh`: `per_session` (default) reuses the first received catalog for a conversation; `per_turn` leaves the supplied catalog unchanged. Other instruction sections and user/tool content remain untouched. Requests without a reliable conversation identity bypass snapshots; shared prompt-cache cohorts are not conversation identities. Snapshots are process-local, expire after four idle hours, and use bounded LRU retention; oversized blocks bypass caching. The dashboard's `src/codex/prompt-layers.ts` and `src/codex/prompt-text-probe.ts` continue observing current files for previews and do not own session snapshots. +Responses request preparation stabilizes incoming `` under `skills.catalog_refresh`: `per_session` (default) reuses the first received catalog for a conversation; `per_turn` leaves the supplied catalog unchanged. Other instruction sections and user/tool content remain untouched. Requests without a reliable conversation identity bypass snapshots; shared prompt-cache cohorts are not conversation identities. Only a body with exactly one catalog block across its instructions and developer/system content takes part; two or more pass through unchanged. A known snapshot is substituted before parsing, but a new catalog is stored only when preparation reaches its success return, so a request rejected by parsing or admission pins nothing. Without a named principal, snapshots are shared by conversation id only on a server that requires no data-plane auth. Snapshots are process-local, expire after four idle hours, and use bounded LRU retention; oversized blocks bypass caching. The dashboard's `src/codex/prompt-layers.ts` and `src/codex/prompt-text-probe.ts` continue observing current files for previews and do not own session snapshots. `/v1/responses` is the main Codex-facing endpoint. The server parses Responses input, routes to a provider, lets the selected adapter speak the upstream protocol, then bridges adapter events back to Responses-compatible streaming output. For an opted-in key-auth provider, a hosted-search continuation stays bound to the API-key selection that served the first leg; the contract is the [hosted-search continuation binding](../providers-and-adapters.md#hosted-search-continuation-binding). diff --git a/tests/responses/responses-skills-snapshot.test.ts b/tests/responses/responses-skills-snapshot.test.ts index b72b51efe2e..a1c8c07fdee 100644 --- a/tests/responses/responses-skills-snapshot.test.ts +++ b/tests/responses/responses-skills-snapshot.test.ts @@ -1,7 +1,7 @@ import { afterEach, beforeEach, describe, expect, test } from "bun:test"; import { getDefaultConfig } from "../../src/config/proxy-env"; import { handleResponses } from "../../src/server/responses/core"; -import { resetSkillsSnapshotCacheForTests } from "../../src/server/responses/skills-snapshot"; +import { resetSkillsSnapshotCacheForTests, resolveSkillsSnapshotScopeKey } from "../../src/server/responses/skills-snapshot"; import type { OcxConfig } from "../../src/types"; import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; @@ -111,3 +111,63 @@ describe("skills catalog snapshots on the Responses request path", () => { .toContain("second-child-catalog"); }); }); + +describe("skills catalog snapshot blockers from the #6027 review", () => { + function body(model: string, developerTexts: string[], extra: Record = {}) { + return JSON.stringify({ + model, stream: false, ...extra, + input: [ + ...developerTexts.map(text => ({ role: "developer", content: [{ type: "input_text", text }] })), + { role: "user", content: "hi" }, + ], + }); + } + async function raw( + config: OcxConfig, model: string, developerTexts: string[], thread: string, extra: Record = {}, + ) { + const response = await handleResponses(new Request("http://localhost/v1/responses", { + method: "POST", + headers: { "content-type": "application/json", "thread-id": thread }, + body: body(model, developerTexts, extra), + }), config, { model: "", provider: "" }); + await response.text(); + return { status: response.status, sent: captured.at(-1) }; + } + const block = (name: string) => `${name}`; + + test("a body with two catalog blocks passes through and pins nothing", async () => { + const config = fixture("openai-responses"); + const two = await raw(config, "fixture/fixture-model", [block("alpha"), block("beta")], "multi"); + expect(two.sent).toContain("alpha"); + expect(two.sent).toContain("beta"); + const one = await raw(config, "fixture/fixture-model", [block("gamma")], "multi"); + expect(one.sent).toContain("gamma"); + // A later two-block body is not rewritten to the stored catalog either. + const again = await raw(config, "fixture/fixture-model", [block("delta"), block("epsilon")], "multi"); + expect(again.sent).toContain("delta"); + expect(again.sent).toContain("epsilon"); + expect(again.sent).not.toContain("gamma"); + }); + + test("a rejected first request pins nothing", async () => { + const config = fixture("openai-responses"); + // The catalog is inspected before parsing; this body then fails parsing (tools must be an array). + const rejected = await raw(config, "fixture/fixture-model", [block("rejected")], "reject-first", { tools: "x" }); + expect(rejected.status).toBe(400); + const first = await raw(config, "fixture/fixture-model", [block("accepted")], "reject-first"); + expect(first.sent).toContain("accepted"); + const second = await raw(config, "fixture/fixture-model", [block("edited")], "reject-first"); + expect(second.sent).toContain("accepted"); + expect(second.sent).not.toContain("edited"); + }); + + test("anonymous callers share only on a server that requires no data-plane auth", () => { + const req = new Request("http://localhost/v1/responses", { headers: { "thread-id": "anon" } }); + const open = { ...getDefaultConfig(), hostname: "127.0.0.1" } as OcxConfig; + const remote = { ...getDefaultConfig(), hostname: "0.0.0.0" } as OcxConfig; + expect(resolveSkillsSnapshotScopeKey({ req, config: open, admission: { kind: "loopback", source: "loopback" } })) + .not.toBeNull(); + expect(resolveSkillsSnapshotScopeKey({ req, config: open })).not.toBeNull(); + expect(resolveSkillsSnapshotScopeKey({ req, config: remote })).toBeNull(); + }); +}); From ad3b374820f7d1ee9f432de432f85cb82cd23081 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:35:36 +0900 Subject: [PATCH 49/75] docs(management-api): describe remote dashboard sessions as they ship The reference still said remote binds never get a session. A trusted Tailscale identity or a pairing grant mints a 12-hour session that each authorized request extends (#2776); other remote operators use the admin token. Found while closing #4055. --- docs-site/src/content/docs/reference/management-api.md | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/docs-site/src/content/docs/reference/management-api.md b/docs-site/src/content/docs/reference/management-api.md index 0a1b76c94e4..464b807780c 100644 --- a/docs-site/src/content/docs/reference/management-api.md +++ b/docs-site/src/content/docs/reference/management-api.md @@ -45,9 +45,12 @@ On a loopback bind, the dashboard bootstrap can receive a short-lived `ocx_sessi Each session lasts five minutes and is bound to the exact dashboard origin. Safe requests must match that origin. Unsafe methods also require the browser `Origin` and the session's CSRF token. -Session issuance is disabled whenever data-plane authentication is required, which includes remote -binds. A remote operator must authenticate with the raw admin token; no loopback-style GUI session -is minted. +When data-plane authentication is required, which includes remote binds, the loopback bootstrap +does not mint a session. A remote dashboard gets a 12-hour session only through a trusted Tailscale +identity (`remoteGui.allowedTailscaleUsers` on the Tailscale management ingress) or a one-use +pairing grant; each authorized request extends it. Otherwise a remote operator authenticates with +the raw admin token, and the dashboard asks for it again after a reload because the session lives +only in page memory. See [Remote hub](../../guides/remote-hub/). ## Common errors From 2256d097e6b4f7fbfdc26eb1d8289a7324241997 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:35:49 +0900 Subject: [PATCH 50/75] docs(management-api): use the site-absolute remote hub link --- docs-site/src/content/docs/reference/management-api.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs-site/src/content/docs/reference/management-api.md b/docs-site/src/content/docs/reference/management-api.md index 464b807780c..f10c36ba19b 100644 --- a/docs-site/src/content/docs/reference/management-api.md +++ b/docs-site/src/content/docs/reference/management-api.md @@ -50,7 +50,7 @@ does not mint a session. A remote dashboard gets a 12-hour session only through identity (`remoteGui.allowedTailscaleUsers` on the Tailscale management ingress) or a one-use pairing grant; each authorized request extends it. Otherwise a remote operator authenticates with the raw admin token, and the dashboard asks for it again after a reload because the session lives -only in page memory. See [Remote hub](../../guides/remote-hub/). +only in page memory. See [Remote hub](/guides/remote-hub/). ## Common errors From bbec11816e322151a02948a9cf021412c78961b0 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 16:39:37 +0900 Subject: [PATCH 51/75] docs(devlog): record train 3 B5 evidence --- .../_plan/260927_merge_train_3/050_batch5.md | 27 +++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/050_batch5.md b/devlog/_plan/260927_merge_train_3/050_batch5.md index 04e8c990816..aaef42bfb9f 100644 --- a/devlog/_plan/260927_merge_train_3/050_batch5.md +++ b/devlog/_plan/260927_merge_train_3/050_batch5.md @@ -26,3 +26,30 @@ Held: #6030 (draft; launchd PATH adoption drops non-PATH changes, two ratchet br trust model. - #5953: the gate reads effective effort after combo overrides; transcript shapes that #5465 does not show stay unprotected, which is today's behavior. + +## Build and evidence + +| Commit | What | +|---|---| +| `4fda15d338` | #5180: canonical key-auth Command Code gets the patient same-key 429 policy; test fails without the fix | +| `3876151d01` | #5953 carried | +| `fb3faab205` | #5953 narrowed: Z.AI host, effective `high`/`max`, checkpoint transcript ≥ 2000 characters, per-field cap raise to 8192; negative tests per boundary | +| `25e01a8e2b` | #6027 carried (layout registries unioned, entry kept on an existing line) | +| `01b7e24d2d` | #6027 blockers: one-block rule, store at the success return, anonymous sharing only on a no-auth server; three tests that fail on the PR head | +| `ad3b374820`, `2256d097e6` | `management-api.md` describes remote dashboard sessions as shipped (#4055) | + +Closed during this cycle with evidence: #3433 (identifiers preserved at the forward boundary, #4365; managed Hermes +sends one, #5742; 26 pinned tests pass). + +Aside: #5953 shows two CHANGES_REQUESTED reviews (the overbreadth this batch narrows); #6027 shows the owner's +three-blocker review this batch answers; #5465, #5569 and #5180 pages captured. One Aside capture failed once with a +daemon `Aside.controlTab` error after an Aside update and succeeded on retry. + +Local proof at `2256d097e6`: typecheck, structure and privacy exit 0; six focused files 60 pass; `tests/adapters/openai` +591 pass; eight request-preparation files 108 pass. The full `tests/responses` directory shows 13 failures in +`responses-compaction-recovery.test.ts` that do not reproduce when that file runs alone (33 pass on this branch and on +`dev`); the same directory run on `dev` is recorded below. + +The full `tests/responses` run on `dev` `4b3737fc5c` shows the same 13 `responses-compaction-recovery` failures +(3567 pass, 13 fail), so they are cross-file interference in a directory run, not this batch; the branch run was 3576 +pass, 13 fail. From 28906a3ec5fdae1d5d4209b15af713d707014d41 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 17:01:04 +0900 Subject: [PATCH 52/75] docs(devlog): plan train 3 B6 --- devlog/_plan/260927_merge_train_3/060_batch6.md | 15 +++++++++++++++ 1 file changed, 15 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/060_batch6.md diff --git a/devlog/_plan/260927_merge_train_3/060_batch6.md b/devlog/_plan/260927_merge_train_3/060_batch6.md new file mode 100644 index 00000000000..fabff72f5f0 --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/060_batch6.md @@ -0,0 +1,15 @@ +# B6 — direct MCP calls in code mode, attested live startup health + +Base: `dev` `7d8459388c` (after B5 #6066). Branch `codex/train3-b6`. + +Previous D (B5): #5953 and #6027 landed with their review fixes; #5180 and the #4055 docs landed. Direction kept. + +| Item | Plan | Review | +|---|---|---| +| #5925 (mdwsk88) | Carry head `06fa8855b` as is. A routed provider's undeclared `mcp____` call is folded into the client's declared custom `exec` code-mode tool instead of failing the turn with a 502. | Kimi review LAND; dedicated security review of the undeclared-tool admission: BLOCKER no (recovery needs a client-declared custom `exec` and an actual routed conversion; arguments stay JSON data; explicit and namespaced declarations win). | +| #5977 (RHODIZSECURITY) | Carry, then close the hold that kept it out of round 2: the server signs a local-read response with `LOCAL_ATTESTATION_PROOF_HEADER` over the request nonce, `fetchBoundLocalManagementRead` verifies it when a caller opts in, and `ocx status` opts in, so a listener that took the port cannot supply a `protected` startup verdict. | Kimi review: the bug is real on dev; the fix reuses the attestation already used by `/healthz` and system restart. | + +Held from this round's reviews, with reasons, for the outcome ledger: #4177 (no loop exists on dev; the rest is a +feature), #4732 (perf rework against `snapshot-select.ts` and measurements needed from the author), #5539 (reverses +test-locked behavior without a reproduction), #4961 (issue withholds a design), #4143 (needs the reporter's desktop +routing details). From bee2ea4357f531ce0c70e8f99ec65f87f2f7a46d Mon Sep 17 00:00:00 2001 From: mdwsk88 <924038395@qq.com> Date: Sun, 27 Sep 2026 17:01:07 +0900 Subject: [PATCH 53/75] fix(responses): route direct mcp tool calls through code-mode exec (#5925) Carried from #5925 into merge train round 3. Co-authored-by: mdwsk88 <924038395@qq.com> --- .../content/docs/guides/codex-integration.md | 9 + scripts/test-layout/layout.json | 1 + src/bridge/response-json.ts | 4 +- src/bridge/sse.ts | 4 +- src/responses/code-mode-helper-compat.ts | 15 +- src/responses/custom-tool-compat.ts | 2 +- .../inference/client-encoder-delivery.ts | 1 + src/server/responses-undeclared-tool-guard.ts | 49 +++- src/server/responses/adapter-delivery.ts | 10 +- src/server/responses/passthrough-delivery.ts | 4 + src/server/responses/passthrough-dispatch.ts | 10 + src/server/responses/run-turn-execution.ts | 6 +- src/types.ts | 1 + src/types/tools.ts | 46 +++- structure/providers-and-adapters.md | 4 + structure/transports/responses-wire-shapes.md | 22 +- .../bridge-legacy-shell-normalization.test.ts | 33 +++ tests/fixtures/test-layout-expected.json | 1 + .../responses-code-mode-mcp-direct.test.ts | 224 ++++++++++++++++++ .../inference-client-encoder-delivery.test.ts | 72 ++++++ 20 files changed, 491 insertions(+), 27 deletions(-) create mode 100644 tests/responses/responses-code-mode-mcp-direct.test.ts diff --git a/docs-site/src/content/docs/guides/codex-integration.md b/docs-site/src/content/docs/guides/codex-integration.md index 1b964b14529..2089cd136e9 100644 --- a/docs-site/src/content/docs/guides/codex-integration.md +++ b/docs-site/src/content/docs/guides/codex-integration.md @@ -653,6 +653,15 @@ code-mode `exec` has the call converted into the matching `tools.(...)` `exec`. A catalog that genuinely declares the bare goal tool keeps it, and a catalog that declares neither the tool nor `exec` still rejects the call as undeclared. +On routed conversions with a verified freeform code-mode `exec` catalog, structured calls +sent directly to +`mcp____` (including a provider-added `default.` prefix) are also +wrapped as nested host-tool calls. This avoids a retry +caused solely by a model omitting the `exec` wrapper. Explicitly declared MCP tools +keep their normal behavior; an ordinary JSON function named `exec` does not enable +this repair. Unknown tools still fail at the host. Tool-call records printed as +ordinary answer text are not executed by this compatibility rule. + For routed Responses turns, an explicit tool-enforcement policy also rejects client tool calls if the request's declared-tool catalog is unavailable. An empty declared catalog rejects every client tool call; Chat and Anthropic clients retain their own tool-validation responsibility. diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index cb4c2759e87..91250a8fd4f 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -1549,6 +1549,7 @@ "responses-bare-echo-helper-fence.test.ts": "responses", "responses-canonical-only-top-level-fields.test.ts": "responses", "responses-code-mode-goal-helpers.test.ts": "responses", + "responses-code-mode-mcp-direct.test.ts": "responses", "responses-code-mode-patch-compile.test.ts": "responses", "responses-code-mode-shell-compile.test.ts": "responses", "responses-compact-handoff-admission.test.ts": "responses", diff --git a/src/bridge/response-json.ts b/src/bridge/response-json.ts index 48fef166e08..c3f57d6ca01 100644 --- a/src/bridge/response-json.ts +++ b/src/bridge/response-json.ts @@ -86,6 +86,8 @@ function buildResponseJSONWithBudget( toolNsMap?: Map; /** Request-visible tool names. Required for client calls when enforcement is explicitly enabled. */ declaredToolNames?: ReadonlySet; + /** Bare custom declarations; unlike freeformToolNames, excludes foreign namespace children. */ + bareCustomToolNames?: ReadonlySet; /** See `bridgeToResponsesSSE`: enforcement is separate from normalization (#4735). */ enforceDeclaredToolNames?: boolean; /** Declared parameter schema per tool name; repairs integral-float integer args (#1611). */ @@ -451,7 +453,7 @@ function buildResponseJSONWithBudget( rememberReasoningForCall(e.id, rawReasoningForNextToolCall, replayCacheScope); } flushToolCall(); - const effectiveName = normalizeDeclaredToolName(e.name, options?.declaredToolNames); + const effectiveName = normalizeDeclaredToolName(e.name, options?.declaredToolNames, undefined, options?.bareCustomToolNames); if ( (options?.enforceDeclaredToolNames === true || options?.declaredToolNames != null) && options?.enforceDeclaredToolNames !== false diff --git a/src/bridge/sse.ts b/src/bridge/sse.ts index c733c1e45d7..730d67dacac 100644 --- a/src/bridge/sse.ts +++ b/src/bridge/sse.ts @@ -101,6 +101,8 @@ export function bridgeToResponsesSSE( onUsage?: (usage: OcxUsage | undefined) => void; /** Request-visible tool names. Required for client calls when enforcement is explicitly enabled. */ declaredToolNames?: ReadonlySet; + /** Bare custom declarations; unlike freeformToolNames, excludes foreign namespace children. */ + bareCustomToolNames?: ReadonlySet; /** * Whether `declaredToolNames` is an authorization boundary this proxy enforces, or only the * catalog used to normalize provider-invented names back to declared ones. @@ -1013,7 +1015,7 @@ export function bridgeToResponsesSSE( rememberReasoningForCall(event.id, rawReasoningForNextToolCall, replayCacheScope); } if (currentToolCall) closeCurrentToolCall(); - const effectiveName = normalizeDeclaredToolName(event.name, options?.declaredToolNames); + const effectiveName = normalizeDeclaredToolName(event.name, options?.declaredToolNames, undefined, options?.bareCustomToolNames); const codeModeHelperName = effectiveName === "exec" && event.name !== effectiveName ? event.name : undefined; diff --git a/src/responses/code-mode-helper-compat.ts b/src/responses/code-mode-helper-compat.ts index 4d4721288ac..6fad7be0c3d 100644 --- a/src/responses/code-mode-helper-compat.ts +++ b/src/responses/code-mode-helper-compat.ts @@ -3,13 +3,16 @@ import { normalizeApplyPatchDelimiters, unwrapFreeformToolInput, } from "./apply-patch-envelope"; -import { declaresCodeModeExec } from "../types/tools"; +import { declaresCodeModeExec, isCodeModeMcpDirectName } from "../types/tools"; import { parseCodeModeShellInput } from "./code-mode-shell-input"; function isPlainObject(value: unknown): value is Record { return !!value && typeof value === "object" && !Array.isArray(value); } +/** A nested host tool reachable as `tools.` when the flattened name is one identifier. */ +const CODE_MODE_IDENTIFIER_NAME = /^[A-Za-z_$][A-Za-z0-9_$]*$/; + /** * Convert a nested Code Mode helper call into unified-exec JavaScript. * @@ -95,6 +98,16 @@ export function compileCodeModeHelperInput( if (helperName === "create_goal" || helperName === "get_goal" || helperName === "update_goal") { return `const result = await tools.${helperName}(${JSON.stringify(args)});\ntext(result);`; } + if (isCodeModeMcpDirectName(helperName)) { + // Direct `mcp____` call under a code-mode catalog: the emitted name IS the + // nested host tool's name, so compile to the same `tools.(args)` the model could + // have written. Dot access when the name is a clean identifier (the common case); bracket + // access otherwise, so a hyphenated server name still addresses the same tool. + const target = CODE_MODE_IDENTIFIER_NAME.test(helperName) + ? `tools.${helperName}` + : `tools[${JSON.stringify(helperName)}]`; + return `const result = await ${target}(${JSON.stringify(args)});\ntext(result);`; + } return `const result = await tools.exec_command(${JSON.stringify(args)});\ntext(result);`; } diff --git a/src/responses/custom-tool-compat.ts b/src/responses/custom-tool-compat.ts index 989af132974..4b15bc9f9ff 100644 --- a/src/responses/custom-tool-compat.ts +++ b/src/responses/custom-tool-compat.ts @@ -76,7 +76,7 @@ export function routedCustomToolTargetName( if (wireName === undefined) return undefined; if (names.has(wireName)) return wireName; if (!isPlainObject(value) || typeof value.namespace === "string") return undefined; - const normalized = normalizeDeclaredToolName(wireName, declaredNames); + const normalized = normalizeDeclaredToolName(wireName, declaredNames, undefined, names); return normalized !== wireName && names.has(normalized) ? normalized : undefined; } diff --git a/src/server/inference/client-encoder-delivery.ts b/src/server/inference/client-encoder-delivery.ts index 1dc2a649e71..5cd6d1c22c2 100644 --- a/src/server/inference/client-encoder-delivery.ts +++ b/src/server/inference/client-encoder-delivery.ts @@ -91,6 +91,7 @@ export interface ClientEncodedDelivery { declaredToolNames?: ReadonlySet; toolParameterSchemas?: ReadonlyMap>; freeformToolNames?: Set; + bareCustomToolNames?: ReadonlySet; toolSearchToolNames?: Set; }; stallTimeoutSec?: number; diff --git a/src/server/responses-undeclared-tool-guard.ts b/src/server/responses-undeclared-tool-guard.ts index 943f19e7d6d..52cc338e509 100644 --- a/src/server/responses-undeclared-tool-guard.ts +++ b/src/server/responses-undeclared-tool-guard.ts @@ -226,6 +226,35 @@ export function collectDeclaredBareWireToolNames(body: unknown): Set { return names; } +/** Custom declarations with no foreign namespace, for code-mode exec recovery. */ +export function collectDeclaredBareCustomWireToolNames(body: unknown): Set { + const names = new Set(); + if (!isPlainObject(body)) return names; + const specGroups: unknown[] = [body.tools]; + if (Array.isArray(body.input)) { + for (const item of body.input) { + if (isPlainObject(item) && (item.type === "additional_tools" || item.type === "tool_search_output")) { + specGroups.push(item.tools); + } + } + } + const add = (tool: unknown): void => { + if (isPlainObject(tool) && tool.type === "custom" && typeof tool.name === "string") names.add(tool.name); + }; + for (const specs of specGroups) { + if (!Array.isArray(specs)) continue; + for (const spec of specs) { + if (!isPlainObject(spec)) continue; + if (spec.type === "namespace") { + if (spec.name === BUILTIN_FUNCTIONS_NAMESPACE && Array.isArray(spec.tools)) { + for (const inner of spec.tools) add(inner); + } + } else add(spec); + } + } + return names; +} + function addNamelessClientCallTypes(callTypes: Set, specs: unknown): void { if (!Array.isArray(specs)) return; for (const spec of specs) { @@ -349,6 +378,7 @@ export function hasExplicitWireToolCatalog(body: unknown): boolean { * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns The undeclared tool call name if unauthorized, or undefined if permitted. */ function undeclaredNameInItem( @@ -357,6 +387,7 @@ function undeclaredNameInItem( declaredNamelessClientCallTypes: ReadonlySet, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string | undefined { if (!isPlainObject(item)) return undefined; if (typeof item.type !== "string") return undefined; @@ -396,7 +427,7 @@ function undeclaredNameInItem( ) return undefined; return name; } - const effectiveName = normalizeDeclaredToolName(name, declared, declaredBare); + const effectiveName = normalizeDeclaredToolName(name, declared, declaredBare, declaredCustom); if (declared.has(effectiveName)) return undefined; return name; } @@ -409,6 +440,7 @@ function undeclaredNameInItem( * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns The name of the first undeclared tool call, or undefined. */ export function undeclaredToolCallName( @@ -417,18 +449,19 @@ export function undeclaredToolCallName( declaredNamelessClientCallTypes: ReadonlySet = EMPTY_DECLARED_NAMELESS_CLIENT_CALL_TYPES, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string | undefined { if (!isPlainObject(payload)) return undefined; if (payload.type === "response.output_item.added" || payload.type === "response.output_item.done") { - return undeclaredNameInItem(payload.item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + return undeclaredNameInItem(payload.item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); } if (payload.type === "response.function_call_arguments.done" && typeof payload.name === "string") { const fakeItem = { type: "function_call", name: payload.name, namespace: payload.namespace }; - return undeclaredNameInItem(fakeItem, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + return undeclaredNameInItem(fakeItem, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); } // Sparse gateways skip incremental items and only ever ship the terminal snapshot. if (payload.type === "response.completed" || payload.type === "response.incomplete") { - return undeclaredToolCallNameInResponse(payload.response, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + return undeclaredToolCallNameInResponse(payload.response, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); } return undefined; } @@ -441,6 +474,7 @@ export function undeclaredToolCallName( * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns The name of the first undeclared tool call, or undefined. */ export function undeclaredToolCallNameInResponse( @@ -449,10 +483,11 @@ export function undeclaredToolCallNameInResponse( declaredNamelessClientCallTypes: ReadonlySet = EMPTY_DECLARED_NAMELESS_CLIENT_CALL_TYPES, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string | undefined { if (!isPlainObject(response) || !Array.isArray(response.output)) return undefined; for (const item of response.output) { - const name = undeclaredNameInItem(item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + const name = undeclaredNameInItem(item, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); if (name !== undefined) return name; } return undefined; @@ -667,6 +702,7 @@ function failedBlocks(name: string, newline: string): readonly string[] { * @param declaredNamelessClientCallTypes - Nameless client call types declared by the request. * @param providerExecutedCallTypes - Call types executed by the provider. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. + * @param declaredCustom - Current bare custom declarations eligible for code-mode recovery. * @returns An SSE block rewrite function. */ export function createUndeclaredToolCallGuardBlockRewrite( @@ -674,6 +710,7 @@ export function createUndeclaredToolCallGuardBlockRewrite( declaredNamelessClientCallTypes: ReadonlySet = EMPTY_DECLARED_NAMELESS_CLIENT_CALL_TYPES, providerExecutedCallTypes: ProviderExecutedCallTypes = EMPTY_PROVIDER_EXECUTED_CALL_TYPES, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): SseBlockRewrite { let tripped = false; return (block: string) => { @@ -686,7 +723,7 @@ export function createUndeclaredToolCallGuardBlockRewrite( } catch { return [block]; } - const name = undeclaredToolCallName(parsed, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare); + const name = undeclaredToolCallName(parsed, declared, declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBare, declaredCustom); if (name !== undefined) { tripped = true; return failedBlocks(name, block.includes("\r\n") ? "\r\n" : "\n"); diff --git a/src/server/responses/adapter-delivery.ts b/src/server/responses/adapter-delivery.ts index 58a445b88e7..321783be985 100644 --- a/src/server/responses/adapter-delivery.ts +++ b/src/server/responses/adapter-delivery.ts @@ -119,7 +119,7 @@ export async function deliverAdapterResponse( continuation: fetchGuardedEmptyCompletionRetry, }) : eventStream; - const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames } = toolBridgeMaps; + const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames } = toolBridgeMaps; // One completion owner for both deliveries: the bridge calls it from its terminal, the // direct client encoder from the fold of the same events. const onCompletedResponse = (response: Record, providerState?: OcxProviderContinuationState) => { @@ -153,7 +153,7 @@ export async function deliverAdapterResponse( fold: { replayCacheScope: parsed._reasoningReplayScope, hideThinkingSummary: parsed.options.hideThinkingSummary, - toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames, + toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames, }, stallTimeoutSec: config.stallTimeoutSec, localUpstream, @@ -176,8 +176,9 @@ export async function deliverAdapterResponse( localUpstream, hideThinkingSummary: parsed.options.hideThinkingSummary, declaredToolNames, + bareCustomToolNames, enforceDeclaredToolNames: options.inboundWire !== "chat" && options.inboundWire !== "anthropic", - toolParameterSchemas, + toolParameterSchemas, ...(options.onFirstOutput ? { onFirstOutput: options.onFirstOutput } : {}), ...(routedCompaction ? { compaction: true } : {}), // Same grok-surface split as the runTurn branch above. @@ -236,7 +237,7 @@ export async function deliverAdapterResponse( } finally { cleanupUpstreamAbort(); } - const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames } = toolBridgeMaps; + const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames } = toolBridgeMaps; let providerState: OcxProviderContinuationState | undefined; const json = buildResponseJSON(events, parsed._responseModelId ?? parsed.modelId, { translatorBudget, @@ -244,6 +245,7 @@ export async function deliverAdapterResponse( hideThinkingSummary: parsed.options.hideThinkingSummary, toolNsMap, declaredToolNames, + bareCustomToolNames, enforceDeclaredToolNames: options.inboundWire !== "chat" && options.inboundWire !== "anthropic", toolParameterSchemas, freeformToolNames, diff --git a/src/server/responses/passthrough-delivery.ts b/src/server/responses/passthrough-delivery.ts index 366c6c00ddc..4c039f0d55c 100644 --- a/src/server/responses/passthrough-delivery.ts +++ b/src/server/responses/passthrough-delivery.ts @@ -313,6 +313,7 @@ export async function deliverPassthroughResponse( | "declaredNamelessClientCallTypes" | "providerExecutedCallTypes" | "declaredBareWireToolNames" + | "recoverableBareCustomWireToolNames" | "rememberPassthroughResponse" | "noteInspectedPayload" | "normalizeFunctionCompletionJson" @@ -336,6 +337,7 @@ export async function deliverPassthroughResponse( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, rememberPassthroughResponse, noteInspectedPayload, normalizeFunctionCompletionJson, @@ -732,6 +734,7 @@ export async function deliverPassthroughResponse( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ) : undefined, grokUpstreamEchoEnabled @@ -998,6 +1001,7 @@ export async function deliverPassthroughResponse( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ); } catch { return undefined; diff --git a/src/server/responses/passthrough-dispatch.ts b/src/server/responses/passthrough-dispatch.ts index 1b953bc0f8f..b78be9e0a97 100644 --- a/src/server/responses/passthrough-dispatch.ts +++ b/src/server/responses/passthrough-dispatch.ts @@ -22,6 +22,7 @@ import { hasExplicitWireToolCatalog, collectDeclaredWireToolNames, collectDeclaredBareWireToolNames, + collectDeclaredBareCustomWireToolNames, collectDeclaredNamelessClientCallTypes, collectProviderExecutedCallTypes, undeclaredToolCallName, @@ -314,6 +315,7 @@ export async function preparePassthroughExchange( const clientExplicitWireToolCatalog = hasExplicitWireToolCatalog(clientToolAuthorizationBody); const clientDeclaredWireToolNames = collectDeclaredWireToolNames(clientToolAuthorizationBody); const clientDeclaredBareWireToolNames = collectDeclaredBareWireToolNames(clientToolAuthorizationBody); + const clientDeclaredBareCustomWireToolNames = collectDeclaredBareCustomWireToolNames(clientToolAuthorizationBody); const clientDeclaredNamelessCallTypes = collectDeclaredNamelessClientCallTypes( clientToolAuthorizationBody, ); @@ -361,6 +363,11 @@ export async function preparePassthroughExchange( ) routedCustomToolRepairNames.add(name); } } + // Only grant direct MCP recovery when this delivery actually restores converted custom + // calls. Native forward and injection paths have no such rewrite. + const recoverableBareCustomWireToolNames = new Set( + [...clientDeclaredBareCustomWireToolNames].filter(name => routedCustomToolNames.has(name)), + ); for (const name of request.convertedRoutedToolSearchNames ?? []) { // The adapter already keeps this set empty when tool_choice forbids the private search. // Its wire name may be collision-aliased, so comparing it to the caller-facing name here @@ -560,6 +567,7 @@ export async function preparePassthroughExchange( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ) !== undefined) { inspectionSawUndeclaredTool = true; } @@ -605,6 +613,7 @@ export async function preparePassthroughExchange( declaredNamelessClientCallTypes, providerExecutedCallTypes, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, ) !== undefined ) { return; @@ -1815,6 +1824,7 @@ export async function preparePassthroughExchange( }, declaredWireToolNames, declaredBareWireToolNames, + recoverableBareCustomWireToolNames, declaredNamelessClientCallTypes, authorizedBareNamespaceToolAliases, normalizeFunctionCompletionJson, diff --git a/src/server/responses/run-turn-execution.ts b/src/server/responses/run-turn-execution.ts index de0dc848117..736be252156 100644 --- a/src/server/responses/run-turn-execution.ts +++ b/src/server/responses/run-turn-execution.ts @@ -551,7 +551,7 @@ export async function executeResponsesRunTurn( return retryQueue.stream(); }; - const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, toolSearchToolNames } = toolBridgeMaps; + const { toolNsMap, declaredToolNames, toolParameterSchemas, freeformToolNames, bareCustomToolNames, toolSearchToolNames } = toolBridgeMaps; const enforceDeclaredToolNames = inboundWire !== "chat" && inboundWire !== "anthropic"; const classifyUndeclaredFirstTool = ( event: AdapterEvent, @@ -559,7 +559,7 @@ export async function executeResponsesRunTurn( if (!enforceDeclaredToolNames || event.type !== "tool_call_start") return undefined; // This tool is declared to the adapter by the private search loop. if (wsPlan && event.name === WEB_SEARCH_TOOL_NAME) return undefined; - const effectiveName = normalizeDeclaredToolName(event.name, declaredToolNames); + const effectiveName = normalizeDeclaredToolName(event.name, declaredToolNames, undefined, bareCustomToolNames); if (declaredToolNames.has(effectiveName)) return undefined; return { type: "error", @@ -673,6 +673,7 @@ export async function executeResponsesRunTurn( stallTimeoutSec, hideThinkingSummary: parsed.options.hideThinkingSummary, declaredToolNames, + bareCustomToolNames, enforceDeclaredToolNames, toolParameterSchemas, ...(options.onFirstOutput ? { onFirstOutput: options.onFirstOutput } : {}), @@ -815,6 +816,7 @@ export async function executeResponsesRunTurn( enforceDeclaredToolNames, toolParameterSchemas, freeformToolNames, + bareCustomToolNames, toolSearchToolNames, ...(routedCompaction ? { compaction: true } : {}), onProviderState: state => { providerState = state; }, diff --git a/src/types.ts b/src/types.ts index 5281666884d..ccca3817fb5 100644 --- a/src/types.ts +++ b/src/types.ts @@ -8,6 +8,7 @@ export { dottedToolName, namespacedToolName, normalizeDeclaredToolName, + isCodeModeMcpDirectName, toolChoiceAliases, createToolChoiceResolver, toolChoiceCandidates, diff --git a/src/types/tools.ts b/src/types/tools.ts index c9066d01b58..0c285ef86f1 100644 --- a/src/types/tools.ts +++ b/src/types/tools.ts @@ -70,8 +70,9 @@ export function dottedToolName(namespace: string | undefined, name: string): str * `apply_patch`, `view_image`, or one of the goal helpers (`create_goal`, `get_goal`, * `update_goal`, #5495) instead of the declared `exec`. Accept these nested helper names only * when the request catalog actually declares `exec` and does not itself declare the emitted name - * (an MCP server may legitimately advertise one under its own namespace). The list is closed: an - * unlisted name is never admitted through `exec`. + * (an MCP server may legitimately advertise one under its own namespace). The helper list itself + * is closed; the one other admission through `exec` is a direct `mcp____` call, + * which `isCodeModeMcpDirectName` recognizes. */ const LEGACY_SHELL_BRIDGE_TOOL_NAMES = ["exec_command", "shell_command"] as const; const CODE_MODE_HELPER_TOOL_NAMES = [ @@ -105,6 +106,23 @@ export const CODE_MODE_HELPER_WIRE_NAMES: ReadonlySet = new Set( CODE_MODE_HELPER_TOOL_NAMES, ); +/** + * A flattened MCP wire name (`mcp____`) emitted as a direct tool call. + * + * Codex code mode reaches the host's nested tools through `tools.(...)` inside `exec`, + * so none of them are declared; routed models (observed: Kimi K3, GLM 5.3) occasionally skip the + * wrapper and call the flattened name directly. Under a code-mode catalog the call is compiled + * into the equivalent `tools.(...)` exec body instead of failing closed — capability- + * equivalent, since the model could have written that JavaScript itself. Server and tool must + * both be non-empty so a bare `mcp__` prefix never qualifies. + */ +export function isCodeModeMcpDirectName(name: string): boolean { + if (!name.startsWith("mcp__")) return false; + const rest = name.slice("mcp__".length); + const separator = rest.indexOf("__"); + return separator > 0 && separator + "__".length < rest.length; +} + /** * Spellings that may never be MANUFACTURED as a bare alias for a namespaced tool. * @@ -137,19 +155,24 @@ export const NAMESPACED_BARE_ALIAS_EXCLUDED_NAMES: ReadonlySet = new Set * The same wrapper may surround an already-flattened namespace identity; accept that exact * declared suffix without treating its child name as a bare declaration. * Also normalizes nested helper names (`exec_command`, `shell_command`, `write_stdin`, - * `apply_patch`, `view_image`, `create_goal`, `get_goal`, `update_goal`) to - * `exec` when code-mode `exec` is declared in the request catalog. + * `apply_patch`, `view_image`, `create_goal`, `get_goal`, `update_goal`) and direct + * `mcp____` calls to `exec` when code-mode `exec` is declared in the + * request catalog. MCP recovery additionally requires explicit custom-tool provenance; + * a structured function named `exec` is not a JavaScript executor. * * @param name - The tool name emitted on the wire by the provider. * @param declared - All wire tool names declared in the request catalog, including aliases. * @param declaredBare - Explicitly declared bare tool names without namespace provenance. * When omitted, falls back to `declared`. + * @param declaredCustom - Custom wire identities from the caller's catalog, never bare aliases + * manufactured from foreign namespaces. * @returns The normalized tool name to expose downstream. */ export function normalizeDeclaredToolName( name: string, declared: ReadonlySet | undefined, declaredBare?: ReadonlySet, + declaredCustom?: ReadonlySet, ): string { if (!declared) return name; if (declared.has(name)) return name; @@ -172,11 +195,12 @@ export function normalizeDeclaredToolName( candidate = bare; } else if ( // Code mode never declares bare helper names; a provider that invents `default.` - // for one still means the nested helper. Strip the prefix so the helper list - // below can rewrite it to `exec` (#4412). + // for one still means the nested helper. The same wrapper can surround a direct MCP + // name, but only a custom exec declaration authorizes that recovery. bare.length > 0 && declared.has(CODE_MODE_EXEC_TOOL_NAME) - && (CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(bare) + && ((CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(bare) + || (declaredCustom?.has(CODE_MODE_EXEC_TOOL_NAME) && isCodeModeMcpDirectName(bare))) && !declared.has("default." + bare) && !declared.has("default__" + bare) ) { @@ -192,7 +216,13 @@ export function normalizeDeclaredToolName( if ((LEGACY_SHELL_BRIDGE_TOOL_NAMES as readonly string[]).some(legacy => declared.has(legacy))) { return candidate; } - return (CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(candidate) + if ((CODE_MODE_HELPER_TOOL_NAMES as readonly string[]).includes(candidate)) { + return CODE_MODE_EXEC_TOOL_NAME; + } + // A direct `mcp____` call names a nested host tool the code-mode catalog + // never declares; `compileCodeModeHelperInput` turns it into the `tools.(...)` + // exec body the model could have written itself. + return declaredCustom?.has(CODE_MODE_EXEC_TOOL_NAME) && isCodeModeMcpDirectName(candidate) ? CODE_MODE_EXEC_TOOL_NAME : candidate; } diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 511f0ce8d26..ad719ac37ec 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -13,6 +13,10 @@ the parser fails the turn before releasing their buffered calls. The capture-only bridge checks each raw tool-use start against the init handshake before buffering; a later init cannot authorize a call that started earlier. +Direct MCP names emitted in a verified custom code-mode catalog follow the +[Responses restoration boundary](transports/responses-wire-shapes.md#direct-mcp-calls-in-code-mode). +Ordinary structured functions named `exec` do not opt into this compatibility path. + RunTurn hosted search uses `src/web-search/run-turn-loop.ts`: synthetic calls remain private, progress reaches the bridge during collection, and a validated terminal precedes search execution. Complete search calls remain actionable at a truncated `done`; cancellation prevents subsequent queries and calls. OAuth preflight replay in `src/server/responses/run-turn-execution.ts` retains the synthetic tool while refreshing credential-scoped route state. In `src/server/responses/sidecar-execution.ts`, a search plan takes priority over image/video bridge execution for both transports; only fetch-capable adapters enter the fetch search loop. Combo preflight allows the private search tool only while a search plan is active; client tool declaration checks and replay-unsafe heartbeat protection remain enforced. diff --git a/structure/transports/responses-wire-shapes.md b/structure/transports/responses-wire-shapes.md index 1c05e1b578a..5205b0eea47 100644 --- a/structure/transports/responses-wire-shapes.md +++ b/structure/transports/responses-wire-shapes.md @@ -1,5 +1,19 @@ # Responses Wire Shapes +## Direct MCP calls in code mode + +On routed bridge or converted-custom passthrough paths, when the request declares a +freeform/custom code-mode `exec`, a structured call to +`mcp____` can be restored as an `exec` call to the matching nested host tool. +The same applies to a provider-added `default.` prefix when neither explicit `default.` nor +`default__` identity was declared. +The request must carry verified custom-tool provenance: an ordinary JSON function named +`exec` does not authorize this repair. Explicitly declared MCP tools keep their identity, +legacy shell catalogs stay unchanged, and unknown nested tools fail at the host. +Names and arguments are serialized as data; plain-text tool-call transcripts are never +promoted into executable calls by this rule. Native forwarding and injection lack this +restoration step, so their undeclared-tool guard still rejects a direct MCP call. + Per-wire request and stream shapes on the Responses data plane: mixed-wire model defaults, xAI agent-message continuation, declared-tool membership by inbound wire, and passthrough SSE stream shapes. The endpoint, dispatch, and credential rules they build on are in @@ -511,9 +525,11 @@ JavaScript. Ordinary JavaScript stays progressive. Coverage: `tests/responses/re An explicit custom-tool denial also requests recovery for unmapped historical results without a live catalog; history never adds current tool authorization. The custom-tool compatibility contract owns lowering and final validation. Muse may wrap an already-flattened namespace identity such as -`default.mcp__server__tool` only when the complete suffix exactly matches a declared namespaced name -and neither explicit `default.` nor `default__` identity exists. It cannot borrow a manufactured bare -alias; unknown suffixes still fail as undeclared tools. See [ADR-0099](../decisions/ADR-0099-responses-http-sse.md). +`default.mcp__server__tool` when the complete suffix exactly matches a declared namespaced name +and neither explicit `default.` nor `default__` identity exists. The custom code-mode `exec` +recovery above is a separate path for undeclared direct MCP names. Neither path can borrow a +manufactured bare alias. Outside code mode, unknown suffixes fail as undeclared tools; inside +code mode, the host rejects unknown nested tools. See [ADR-0099](../decisions/ADR-0099-responses-http-sse.md). > Decision record: [ADR-0099](../decisions/ADR-0099-responses-http-sse.md) diff --git a/tests/adapters/bridge-legacy-shell-normalization.test.ts b/tests/adapters/bridge-legacy-shell-normalization.test.ts index cc16fdec090..1530435dd50 100644 --- a/tests/adapters/bridge-legacy-shell-normalization.test.ts +++ b/tests/adapters/bridge-legacy-shell-normalization.test.ts @@ -160,4 +160,37 @@ describe("bridge normalizes code-mode helper names against the declared catalog" expect(sse).not.toContain("tools.view_image"); expect(sse).not.toContain('"name":"exec"'); }); + + // Codex Desktop code mode declares only `exec`; routed models (observed: Kimi K3, GLM 5.3) + // sometimes call a nested host tool by its flattened `mcp____` name directly. + // The guard failed those turns closed, which the desktop surfaced as reconnect banners. + test("a direct mcp tool call is delivered as exec running the nested host call", async () => { + const sse = await drain(bridgeToResponsesSSE( + toolTurn("mcp__codex_app__get_usage_limits", "{}"), + "fixture-model", + undefined, + new Set(["exec"]), + undefined, + undefined, + 50_000, + { declaredToolNames: new Set(["exec"]), bareCustomToolNames: new Set(["exec"]) }, + )); + expect(sse).not.toContain("undeclared client tool"); + expect(sse).toContain('"name":"exec"'); + expect(sse).toContain('await tools.mcp__codex_app__get_usage_limits({})'); + }); + + test("a malformed mcp-prefixed name still fails the turn", async () => { + const sse = await drain(bridgeToResponsesSSE( + toolTurn("mcp__solo", "{}"), + "fixture-model", + undefined, + new Set(["exec"]), + undefined, + undefined, + 50_000, + { declaredToolNames: new Set(["exec"]), bareCustomToolNames: new Set(["exec"]) }, + )); + expect(sse).toContain("undeclared client tool"); + }); }); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 89dfe34703d..f53b8b96c4a 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1393,6 +1393,7 @@ "responses-bare-echo-helper-fence.test.ts": "responses", "responses-canonical-only-top-level-fields.test.ts": "responses", "responses-code-mode-goal-helpers.test.ts": "responses", + "responses-code-mode-mcp-direct.test.ts": "responses", "responses-code-mode-patch-compile.test.ts": "responses", "responses-code-mode-shell-compile.test.ts": "responses", "responses-compact-handoff-admission.test.ts": "responses", diff --git a/tests/responses/responses-code-mode-mcp-direct.test.ts b/tests/responses/responses-code-mode-mcp-direct.test.ts new file mode 100644 index 00000000000..d1d0cb34f87 --- /dev/null +++ b/tests/responses/responses-code-mode-mcp-direct.test.ts @@ -0,0 +1,224 @@ +import { describe, expect, test } from "bun:test"; +import { bridgeToResponsesSSE, buildResponseJSON } from "../../src/bridge"; +import { restoreRoutedCustomCallsInJson, rewriteRoutedCustomToolsForUpstream } from "../../src/responses/custom-tool-compat"; +import { compileCodeModeHelperInput } from "../../src/responses/code-mode-helper-compat"; +import { parseRequest } from "../../src/responses/parser"; +import { buildToolBridgeMaps } from "../../src/server/responses"; +import { createRoutedCustomToolRestoreBlockRewrite } from "../../src/server/responses-custom-tool-repair"; +import { + collectDeclaredBareCustomWireToolNames, + currentTurnWireToolCatalogBody, + undeclaredToolCallNameInResponse, +} from "../../src/server/responses-undeclared-tool-guard"; +import type { AdapterEvent } from "../../src/types"; +import { isCodeModeMcpDirectName, normalizeDeclaredToolName } from "../../src/types/tools"; +import { dataPayload, frame } from "../helpers/custom-tool-repair-fixtures"; + +const CODE_MODE = new Set(["exec"]); +const MCP_NAME = "mcp__codex_app__get_usage_limits"; +const CALL = { type: "function_call", id: "fc_mcp", call_id: "call_mcp", name: MCP_NAME, arguments: "{}" }; +const CUSTOM_EXEC = { type: "namespace", name: "functions", tools: [{ type: "custom", name: "exec", format: { type: "text" } }] }; +const FUNCTION_EXEC = { type: "namespace", name: "functions", tools: [{ type: "function", name: "exec", parameters: { type: "object" } }] }; +const FOREIGN_EXEC = { type: "namespace", name: "mcp__remote", tools: [{ type: "custom", name: "exec", format: { type: "text" } }] }; + +function mapsFor(tools: unknown[]) { + const parsed = parseRequest({ model: "test-model", input: "run it", tools }); + return buildToolBridgeMaps(parsed); +} + +function eventsFor(name: string): AdapterEvent[] { + return [ + { type: "tool_call_start", id: "call_mcp", name }, + { type: "tool_call_delta", id: "call_mcp", arguments: "{}" }, + { type: "tool_call_end", id: "call_mcp" }, + { type: "done" }, + ]; +} + +// Codex Desktop code mode declares only the freeform `exec` shell; every nested host tool +// (`tools.mcp__codex_app__get_usage_limits`, ...) is reachable through it but undeclared. +// Routed models (observed: Kimi K3, GLM 5.3) sometimes call the flattened MCP name directly +// instead of wrapping it in exec JavaScript, and the undeclared-tool guard failed those turns +// closed — visible in the desktop client as "reconnecting N/5" banners. These pin the +// normalization that compiles such a call into the exec body the model could have written. +describe("code-mode direct mcp tool-call recovery", () => { + test("recognizes only well-formed flattened mcp names", () => { + expect(isCodeModeMcpDirectName("mcp__codex_app__get_usage_limits")).toBe(true); + expect(isCodeModeMcpDirectName("mcp__my-server__do_thing")).toBe(true); + for (const name of ["mcp__", "mcp__server", "mcp____tool", "mcp__server__", "exec_command", "mcp_x__y"]) { + expect(isCodeModeMcpDirectName(name)).toBe(false); + } + }); + + test("maps a direct mcp call only through an explicitly custom exec", () => { + expect(normalizeDeclaredToolName(MCP_NAME, CODE_MODE, undefined, CODE_MODE)).toBe("exec"); + expect(normalizeDeclaredToolName(`default.${MCP_NAME}`, CODE_MODE, undefined, CODE_MODE)).toBe("exec"); + expect(normalizeDeclaredToolName(MCP_NAME, CODE_MODE)).toBe(MCP_NAME); + expect(normalizeDeclaredToolName(`default.${MCP_NAME}`, CODE_MODE)).toBe(`default.${MCP_NAME}`); + expect(normalizeDeclaredToolName("mcp__codex_app__get_usage_limits", new Set())).toBe("mcp__codex_app__get_usage_limits"); + // A catalog that declares the name itself keeps the call's own identity. + expect(normalizeDeclaredToolName( + "mcp__codex_app__get_usage_limits", + new Set(["exec", MCP_NAME]), undefined, CODE_MODE, + )).toBe("mcp__codex_app__get_usage_limits"); + // The flat-bridge shape (legacy shell names declared next to exec) is not code mode. + expect(normalizeDeclaredToolName( + "mcp__codex_app__get_usage_limits", + new Set(["exec", "exec_command"]), undefined, CODE_MODE, + )).toBe("mcp__codex_app__get_usage_limits"); + // Malformed mcp-ish names stay undeclared. + expect(normalizeDeclaredToolName("mcp__server", CODE_MODE, undefined, CODE_MODE)).toBe("mcp__server"); + expect(normalizeDeclaredToolName(`default.${MCP_NAME}`, new Set(["exec", `default.${MCP_NAME}`]), undefined, CODE_MODE)) + .toBe(`default.${MCP_NAME}`); + }); + + test("catalog provenance excludes JSON functions and foreign namespace aliases", () => { + const custom = mapsFor([CUSTOM_EXEC]); + const ordinary = mapsFor([FUNCTION_EXEC]); + const foreign = mapsFor([FOREIGN_EXEC]); + expect(custom.bareCustomToolNames).toEqual(CODE_MODE); + expect(ordinary.declaredToolNames.has("exec")).toBe(true); + expect(ordinary.bareCustomToolNames.has("exec")).toBe(false); + expect(foreign.declaredToolNames.has("exec")).toBe(false); + expect(foreign.bareCustomToolNames.has("exec")).toBe(false); + expect(foreign.declaredToolNames.has("mcp__remote__exec")).toBe(true); + for (const maps of [ordinary, foreign]) { + expect(normalizeDeclaredToolName(MCP_NAME, maps.declaredToolNames, undefined, maps.bareCustomToolNames)).toBe(MCP_NAME); + } + }); + + test("an explicitly declared MCP function keeps its namespaced identity", () => { + const maps = mapsFor([CUSTOM_EXEC, { + type: "namespace", name: "mcp__codex_app", + tools: [{ type: "function", name: "get_usage_limits", parameters: { type: "object" } }], + }]); + expect(buildResponseJSON(eventsFor(MCP_NAME), "fixture", { + ...maps, enforceDeclaredToolNames: true, + }).output).toMatchObject([{ + type: "function_call", name: "get_usage_limits", namespace: "mcp__codex_app", arguments: "{}", + }]); + }); + + test("compiles the call to the matching nested host tool", () => { + expect(compileCodeModeHelperInput('{"limit":3}', "mcp__codex_app__list_threads")).toBe( + 'const result = await tools.mcp__codex_app__list_threads({"limit":3});\ntext(result);', + ); + // A hyphenated name is not one identifier: bracket access addresses the same tool. + expect(compileCodeModeHelperInput("{}", "mcp__my-server__do_thing")).toBe( + 'const result = await tools["mcp__my-server__do_thing"]({});\ntext(result);', + ); + // Malformed provider text stays data; nested-tool validation rejects it, not JavaScript. + expect(compileCodeModeHelperInput("not json", "mcp__x__y")).toBe( + 'const result = await tools.mcp__x__y("not json");\ntext(result);', + ); + expect(compileCodeModeHelperInput("{}", `default.${MCP_NAME}`)).toBe( + `const result = await tools.${MCP_NAME}({});\ntext(result);`, + ); + }); + + test("generated JavaScript uses host tool lookup and keeps arguments as data", async () => { + const args = { query: '"; throw new Error("injected") //', limit: 3 }; + const input = compileCodeModeHelperInput(JSON.stringify(args), MCP_NAME); + const run = new Function("tools", "text", `return (async () => { ${input} })();`); + const calls: unknown[] = []; + const output: unknown[] = []; + await run({ [MCP_NAME]: async (value: unknown) => { calls.push(value); return "ok"; } }, (value: unknown) => output.push(value)); + expect(calls).toEqual([args]); + expect(output).toEqual(["ok"]); + await expect(run({}, () => {})).rejects.toThrow(); + }); + + test("guard admits direct calls only with current bare custom provenance", () => { + const source = { output: [CALL] }; + expect(undeclaredToolCallNameInResponse(source, CODE_MODE, undefined, undefined, undefined, CODE_MODE)).toBeUndefined(); + expect(undeclaredToolCallNameInResponse( + { output: [{ ...CALL, name: `default.${MCP_NAME}` }] }, CODE_MODE, + undefined, undefined, undefined, CODE_MODE, + )).toBeUndefined(); + expect(undeclaredToolCallNameInResponse(source, CODE_MODE)).toBe(MCP_NAME); + expect(undeclaredToolCallNameInResponse(source, new Set())).toBe(MCP_NAME); + const current = { tools: [FUNCTION_EXEC], input: [{ type: "additional_tools", tools: [CUSTOM_EXEC] }] }; + expect(collectDeclaredBareCustomWireToolNames(current)).toEqual(CODE_MODE); + expect(collectDeclaredBareCustomWireToolNames(currentTurnWireToolCatalogBody(current, 1))).toEqual(new Set()); + expect(collectDeclaredBareCustomWireToolNames({ tools: [FOREIGN_EXEC] })).toEqual(new Set()); + expect(undeclaredToolCallNameInResponse( + { output: [{ ...CALL, name: "get_usage_limits", namespace: "mcp__codex_app" }] }, + CODE_MODE, undefined, undefined, undefined, CODE_MODE, + )).toBe("get_usage_limits"); + // A name that only looks mcp-ish is still blocked under a code-mode catalog. + const malformed = { + output: [{ type: "function_call", id: "fc_x", call_id: "call_x", name: "mcp__solo", arguments: "{}" }], + }; + expect(undeclaredToolCallNameInResponse(malformed, CODE_MODE, undefined, undefined, undefined, CODE_MODE)).toBe("mcp__solo"); + }); + + test("restores a recorded direct mcp call as the declared exec", () => { + const source = { output: [CALL] }; + const restored = JSON.parse(restoreRoutedCustomCallsInJson( + JSON.stringify(source), + CODE_MODE, + new Set(), + CODE_MODE, + )); + expect(restored.output).toMatchObject([{ + type: "custom_tool_call", + name: "exec", + call_id: "call_mcp", + input: 'const result = await tools.mcp__codex_app__get_usage_limits({});\ntext(result);', + }]); + expect(undeclaredToolCallNameInResponse(restored, CODE_MODE)).toBeUndefined(); + }); + + test("native SSE restoration compiles only a converted custom exec call", () => { + for (const name of [MCP_NAME, `default.${MCP_NAME}`]) { + const customNames = rewriteRoutedCustomToolsForUpstream({ tools: [CUSTOM_EXEC] }, false).names; + const rewrite = createRoutedCustomToolRestoreBlockRewrite(customNames, undefined, new Set(), CODE_MODE); + const added = rewrite(frame("response.output_item.added", { + output_index: 0, + item: { ...CALL, id: "fc_native", name, arguments: "", status: "in_progress" }, + })); + expect(dataPayload(added[0]!).item).toMatchObject({ type: "custom_tool_call", name: "exec" }); + const done = rewrite(frame("response.function_call_arguments.done", { + output_index: 0, item_id: "fc_native", arguments: "{}", + })); + expect(dataPayload(done[0]!).input).toBe(`const result = await tools.${MCP_NAME}({});\ntext(result);`); + rewrite.dispose?.(); + } + }); + + test("JSON, SSE and historical restoration agree on custom, ordinary and foreign exec", async () => { + for (const [tools, accepted] of [ + [[CUSTOM_EXEC], true], + [[FUNCTION_EXEC], false], + [[FOREIGN_EXEC], false], + ] as const) { + const maps = mapsFor([...tools]); + const options = { ...maps, enforceDeclaredToolNames: true }; + for (const name of [MCP_NAME, `default.${MCP_NAME}`]) { + const expectedInput = `const result = await tools.${MCP_NAME}({});\ntext(result);`; + const json = buildResponseJSON(eventsFor(name), "fixture", options); + async function* streamEvents(): AsyncGenerator { yield* eventsFor(name); } + const stream = bridgeToResponsesSSE( + streamEvents(), "fixture", maps.toolNsMap, maps.freeformToolNames, + maps.toolSearchToolNames, undefined, 50_000, options, + ); + const text = await new Response(stream).text(); + const payloads = text.split(/\r?\n\r?\n/).filter(block => block.includes("data: {")).map(dataPayload); + const historical = JSON.parse(restoreRoutedCustomCallsInJson( + JSON.stringify({ output: [{ ...CALL, name }] }), + rewriteRoutedCustomToolsForUpstream({ tools }, false).names, + new Set(), maps.declaredToolNames, + )) as { output: Array> }; + if (accepted) { + expect(json.output).toMatchObject([{ type: "custom_tool_call", name: "exec", input: expectedInput }]); + expect(payloads.find(p => p.type === "response.custom_tool_call_input.done")?.input).toBe(expectedInput); + expect(historical.output[0]).toMatchObject({ type: "custom_tool_call", name: "exec", input: expectedInput }); + } else { + expect(json.status).toBe("failed"); + expect(payloads.some(p => p.type === "response.failed")).toBe(true); + expect(historical.output[0]).toMatchObject({ type: "function_call", name }); + } + } + } + }); +}); diff --git a/tests/server/inference-client-encoder-delivery.test.ts b/tests/server/inference-client-encoder-delivery.test.ts index f2aa53baaca..e11a4d9dff9 100644 --- a/tests/server/inference-client-encoder-delivery.test.ts +++ b/tests/server/inference-client-encoder-delivery.test.ts @@ -16,6 +16,9 @@ import { beginInferenceAttempt } from "../../src/server/inference/attempt"; import { responseWithDeferredRequestLog } from "../../src/server/relay"; import type { RequestLogContext, RequestLogEntry } from "../../src/server/request-log"; import { markProtocolEntry, protocolTraceForRequest } from "../../src/protocols/trace"; +import { parseRequest } from "../../src/responses/parser"; +import { buildToolBridgeMaps } from "../../src/server/responses"; +import { deliverAdapterResponse } from "../../src/server/responses/adapter-delivery"; import type { AdapterEvent, OcxConfig, OcxUsage } from "../../src/types"; import { createTestTranslatorBudget } from "../helpers/translator-budget"; @@ -184,4 +187,73 @@ describe("deliverClientEncodedResponse", () => { expect(body.choices[0]!.message.content).toBe("hi"); expect(run.completed).toHaveLength(1); }); + +}); + +// The routed-adapter branch assembles the encoder's fold itself. Direct MCP recovery reaches the +// completed response only when that fold carries the request's custom exec provenance, so this +// drives the real delivery entry point rather than a hand-built fold. +describe("deliverAdapterResponse through a client encoder", () => { + type Args = Parameters; + + test("the fold keeps code-mode direct MCP recovery", async () => { + const parsed = parseRequest({ + model: "internal/model", + input: "check usage", + stream: true, + store: false, + tools: [{ type: "namespace", name: "functions", tools: [{ type: "custom", name: "exec", format: { type: "text" } }] }], + }); + const toolBridgeMaps = buildToolBridgeMaps(parsed); + expect(toolBridgeMaps.bareCustomToolNames).toEqual(new Set(["exec"])); + const toolEvents: AdapterEvent[] = [ + { type: "tool_call_start", id: "call_mcp", name: "mcp__codex_app__get_usage_limits" }, + { type: "tool_call_delta", id: "call_mcp", arguments: "{}" }, + { type: "tool_call_end", id: "call_mcp" }, + { type: "done" }, + ]; + const completed: Record[] = []; + const response = await deliverAdapterResponse( + { + logCtx: { model: "m", provider: "p" }, + options: { clientEncoder: { protocol: "chat", stream: false, model: "client-model" } }, + config: {}, + } as unknown as Args[0], + { + parsed, + translatorBudget: createTestTranslatorBudget(), + toolBridgeMaps, + rememberKiroDeliveredFinalAnswer: () => {}, + responseStateOptions: () => ({}), + } as unknown as Args[1], + { + activeAdapter: { name: "anthropic", parseStream: () => replay(toolEvents) }, + bindKeyUsageFromBridge: () => {}, + } as unknown as Args[2], + { routedCompaction: undefined } as unknown as Args[3], + { + cancelResponseCompletion: () => {}, + commitReasoningReplayServingRoute: () => {}, + continuationStateForResponse: () => undefined, + notifyResponseComplete: (folded: Record) => { completed.push(folded); }, + } as unknown as Args[4], + { emptyCompletionGuardEnabled: false }, + { + upstreamResponse: new Response(""), + upstream: new AbortController(), + cleanupUpstreamAbort: () => {}, + localUpstream: false, + } as unknown as Args[6], + { terminalGuardEnabled: false } as unknown as Args[7], + ); + expect(response.status).toBe(200); + expect(completed).toHaveLength(1); + expect(completed[0]).toMatchObject({ + output: [{ + type: "custom_tool_call", + name: "exec", + input: 'const result = await tools.mcp__codex_app__get_usage_limits({});\ntext(result);', + }], + }); + }); }); From 2a383cbc1f99b5e8b5edb2c06eeb9098f3325ea1 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 17:01:16 +0900 Subject: [PATCH 54/75] chore(integration): keep the #5925 layout entry on a shared line --- scripts/test-layout/layout.json | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 91250a8fd4f..7fcccbcc734 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -173,7 +173,7 @@ "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", "restart-replacement.test.ts": "server", "deepseek-artifact-tool-schema.test.ts": "providers", - "client-config-export-output-limit.test.ts": "config", "responses-skills-snapshot.test.ts": "responses", + "client-config-export-output-limit.test.ts": "config", "responses-skills-snapshot.test.ts": "responses", "responses-code-mode-mcp-direct.test.ts": "responses", "openai-chat-serialized-tool-call-scaling.test.ts": "adapters/openai", "openai-chat-tool-call-id-remint.test.ts": "adapters/openai", "coding-agent-json-lines-scaling.test.ts": "providers", @@ -1549,7 +1549,6 @@ "responses-bare-echo-helper-fence.test.ts": "responses", "responses-canonical-only-top-level-fields.test.ts": "responses", "responses-code-mode-goal-helpers.test.ts": "responses", - "responses-code-mode-mcp-direct.test.ts": "responses", "responses-code-mode-patch-compile.test.ts": "responses", "responses-code-mode-shell-compile.test.ts": "responses", "responses-compact-handoff-admission.test.ts": "responses", From f2727befa81c8c0efbbdcd5fb24018b912325918 Mon Sep 17 00:00:00 2001 From: RHODIZSECURITY <180237049+RHODIZSECURITY@users.noreply.github.com> Date: Sun, 27 Sep 2026 17:01:21 +0900 Subject: [PATCH 55/75] fix(status): trust attested live startup health (#5977) Carried from #5977 into merge train round 3. Co-authored-by: RHODIZSECURITY <180237049+RHODIZSECURITY@users.noreply.github.com> --- .../content/docs/reference/cli/lifecycle.md | 9 ++ src/cli/doctor.ts | 17 +-- src/cli/status.ts | 93 +++++++++++- tests/cli/cli-status-startup-health.test.ts | 133 ++++++++++++++++++ 4 files changed, 237 insertions(+), 15 deletions(-) create mode 100644 tests/cli/cli-status-startup-health.test.ts diff --git a/docs-site/src/content/docs/reference/cli/lifecycle.md b/docs-site/src/content/docs/reference/cli/lifecycle.md index 8f66ede7b01..da90092f9c5 100644 --- a/docs-site/src/content/docs/reference/cli/lifecycle.md +++ b/docs-site/src/content/docs/reference/cli/lifecycle.md @@ -185,6 +185,15 @@ opencodex wrote names a local port the running proxy does not serve. `ocx sync` immediately, and a running owner proxy may also re-point it on its own once nothing answers on that port. A sibling's status does not report it, because its routing names the proxy it runs beside. +When a live proxy has already passed the identity/liveness check, `ocx status` prefers that process's +attested startup-health report for restart safety and service viability. This avoids false negatives +from a shell-local service-manager probe that lacks the running service's manager environment. The +live report is schema-validated; if it is unavailable or malformed, status falls back to the local +service and shim diagnostics. `ocx doctor` uses the same live-first rule for its **Codex restart safety** +section, so the two commands should agree on restart protection. If you are diagnosing a discrepancy, +compare the reported live startup verdict with the local service details rather than treating the shell +probe as more authoritative. + Human output also includes an **OAuth health** block after the OAuth logins summary: `OAuth health: ok` when every known account is healthy, or `OAuth health: warning` with one redacted line per non-healthy account (provider, masked account id, status such as reauthentication required, rate or diff --git a/src/cli/doctor.ts b/src/cli/doctor.ts index 42482a2d9cf..01b97e6af0c 100644 --- a/src/cli/doctor.ts +++ b/src/cli/doctor.ts @@ -12,7 +12,7 @@ import { homedir } from "node:os"; import { dirname, join } from "node:path"; import { getConfigDir, getConfigPath, readConfigDiagnostics } from "../config"; import { readPid } from "../config/process-state"; -import { probeUncleanExitState } from "./status"; +import { fetchLiveStartupHealth, probeUncleanExitState, selectStatusStartupHealth } from "./status"; import { findLiveProxy, probeHostname, type LiveProxy } from "../server/proxy-liveness"; import { directLocalHttpFetch } from "../server/direct-local-http"; import { BUN_RUNTIME_SOURCES } from "../lib/bun-runtime"; @@ -1354,7 +1354,14 @@ export async function runDoctor(args: string[] = []): Promise { diagnoseCodexShim(), serviceTokenPresent, ); - const startup = collectStartupHealth(doctorConfig); + // Use the same attested live startup verdict as `ocx status` when the proxy is already + // identity-verified. A shell-local systemd probe can be a false negative for a system-wide + // service because the shell does not inherit the manager-owned environment. + const live = await findLiveProxy({ + configFn: () => ({ port: doctorConfig.port, hostname: doctorConfig.hostname }), + }); + const liveStartup = live ? await fetchLiveStartupHealth(live) : null; + const startup = selectStatusStartupHealth(liveStartup, () => collectStartupHealth(doctorConfig)); console.log("\nCodex restart safety"); console.log(` ${startup.rebootSafe ? "ok " : "!! "} ${startupHealthSummary(startup)}`); console.log(` ${formatStartupRoutingDetail(startup)}`); @@ -1393,12 +1400,6 @@ export async function runDoctor(args: string[] = []): Promise { } } - // #618: identity-verified liveness first so pid-file absence does not hide a live service. - // Reuse the diagnostics config already loaded above so doctor stays read-only on malformed JSON. - const live = await findLiveProxy({ - configFn: () => ({ port: doctorConfig.port, hostname: doctorConfig.hostname }), - }); - // Mirrors `ocx status` through the same comparison rather than a second implementation: // two diagnostics disagreeing about whether an install is stale is worse than one (#2701). // No extra probe -- findLiveProxy already carried the version back. diff --git a/src/cli/status.ts b/src/cli/status.ts index fdc5a202c34..fd55aaecd4f 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -31,6 +31,8 @@ import { tokenCollidesWithAdmin } from "../lib/admin-secrets"; export { proxyHealthFailureReason, isConnectionRefused, isUncleanExitEvidence, probeUncleanExitState } from "./status-probes"; export type { ListenTarget } from "./status-probes"; import { checkProxyHealth, probeUncleanExitState, type ListenTarget } from "./status-probes"; +import { LOCAL_MANAGEMENT_READ_PATHS } from "../lib/local-management-capability"; +import { fetchBoundLocalManagementRead } from "../server/local-management-read-client"; /** * The state of the data-plane admission secret the SERVICE will use. State only -- never the value. @@ -235,6 +237,84 @@ function statusDashboardUrl(config: StatusListenConfig, hostname: string | undef return `http://${dashboardHostname}:${port}/`; } +const STARTUP_HEALTH_BOOLEAN_FIELDS = [ + "routingInjected", "localRoutingDependency", "autostartEnabled", "rebootSafe", + "serviceInstalled", "serviceViable", "serviceEnabled", "serviceRunning", + "serviceStale", "serviceConflict", "shimInstalled", "shimHealthy", + "serviceSupported", "diagnosticStale", +] as const; + +export async function fetchLiveStartupHealth( + live: NonNullable>>, + deps: Parameters[2] = {}, +): Promise { + const result = await fetchBoundLocalManagementRead( + live, LOCAL_MANAGEMENT_READ_PATHS.startupHealth, { timeoutMs: 1_500, ...deps }, + ); + if (result.kind !== "response" || !result.response.ok) return null; + let payload: unknown; + try { payload = await result.response.json(); } catch { return null; } + if (!payload || typeof payload !== "object" || Array.isArray(payload)) return null; + const row = payload as Record; + if (row.status !== "native" && row.status !== "protected" && row.status !== "at-risk") return null; + if (row.protection !== "service" && row.protection !== "shim" && row.protection !== "none") return null; + if (row.routingKind !== "native" && row.routingKind !== "opencodex-local" + && row.routingKind !== "custom-local" && row.routingKind !== "custom-remote" && row.routingKind !== "unknown") return null; + if (row.shimCoverage !== "full" && row.shimCoverage !== "cli-only" && row.shimCoverage !== "none") return null; + if (typeof row.platform !== "string") return null; + if (row.recommendedCommand !== null && typeof row.recommendedCommand !== "string") return null; + if (!row.commands || typeof row.commands !== "object" || Array.isArray(row.commands)) return null; + for (const key of ["installService", "repairService", "installShim", "restoreNative"] as const) { + if (typeof (row.commands as Record)[key] !== "string") return null; + } + if (row.routingAdoption !== undefined) { + if (!row.routingAdoption || typeof row.routingAdoption !== "object" || Array.isArray(row.routingAdoption)) return null; + const adoption = row.routingAdoption as Record; + if (adoption.adoption !== "not-applicable" && adoption.adoption !== "adopted" + && adoption.adoption !== "pending-client-restart" && adoption.adoption !== "unknown") return null; + if (adoption.injectedAtMs !== null && typeof adoption.injectedAtMs !== "number") return null; + if (typeof adoption.observedClients !== "number" || !Array.isArray(adoption.staleClients)) return null; + for (const client of adoption.staleClients) { + if (!client || typeof client !== "object" || Array.isArray(client)) return null; + const row = client as Record; + if (typeof row.pid !== "number" || typeof row.startedAtMs !== "number") return null; + } + } + for (const key of STARTUP_HEALTH_BOOLEAN_FIELDS) if (typeof row[key] !== "boolean") return null; + return payload as StartupHealth; +} + +/** Prefer an attested live verdict and evaluate the local fallback only when live state is absent. */ +export function selectStatusStartupHealth( + liveStartup: StartupHealth | null, + fallback: () => StartupHealth, +): StartupHealth { + return liveStartup ?? fallback(); +} + +/** Build the service summary from the same startup source that `ocx status` selected. */ +export function statusServiceSummary( + liveStartup: StartupHealth | null, + service: Pick, "installed" | "summary">, + live: boolean, +): string { + if (liveStartup) { + if (liveStartup.protection === "service" && liveStartup.serviceViable) { + return `running under the live managed service (logs: ${serviceLogPath()})`; + } + const state = [ + liveStartup.serviceInstalled ? "installed" : "absent", + liveStartup.serviceRunning ? "running" : "not running", + liveStartup.serviceViable ? "viable" : "not viable", + ].join(", "); + const action = liveStartup.recommendedCommand ? `; run '${liveStartup.recommendedCommand}'` : ""; + return `live startup reports service ${state}${action} (logs: ${serviceLogPath()})`; + } + return service.installed && !live + ? `${service.summary} — registered but NOT serving; see ${serviceLogPath()} and re-run 'ocx service repair'` + : service.summary; +} + /** * The hub block, or null when this machine is not a hub. * @@ -666,20 +746,19 @@ export async function collectStatus(): Promise { hostname: config.hostname, }); const bunRuntime = durableBunRuntime(); + const liveStartup = live ? await fetchLiveStartupHealth(live) : null; const service = diagnoseService(); // A service can be registered and still not serve: the manager reports the job - // either way. `live` was already identity-probed a few lines above, so cross-check - // rather than print registration as if it were service. - const serviceSummary = service.installed && !live - ? `${service.summary} — registered but NOT serving; see ${serviceLogPath()} and re-run 'ocx service repair'` - : service.summary; + // either way. When the identity-probed live proxy provides an attested startup verdict, + // prefer it over a shell-local service-manager probe that lacks the service environment. + const serviceSummary = statusServiceSummary(liveStartup, service, Boolean(live)); const codexShim = diagnoseCodexShim(); const codexShimSummary = codexShim.summary; - const startup = collectStartupHealth(config, { + const startup = selectStatusStartupHealth(liveStartup, () => collectStartupHealth(config, { service, shim: codexShim, routingKind: getCodexRoutingKind(), - }); + })); const codexPlugins = diagnoseCodexBundledPlugins(); const lastClamp = loadLastEffortClamp(); const clampActive = effortClampAppliesToRuntime(lastClamp, resolvedRuntime.runtime); diff --git a/tests/cli/cli-status-startup-health.test.ts b/tests/cli/cli-status-startup-health.test.ts new file mode 100644 index 00000000000..fd51366b7ad --- /dev/null +++ b/tests/cli/cli-status-startup-health.test.ts @@ -0,0 +1,133 @@ +import { describe, expect, test } from "bun:test"; +import { fetchLiveStartupHealth, selectStatusStartupHealth, statusServiceSummary } from "../../src/cli/status"; +import type { StartupHealth } from "../../src/codex/autostart-health"; + +const LIVE = { + pid: 4242, + port: 10101, + hostname: "127.0.0.1", + source: "runtime" as const, +}; + +const SECRET = "A".repeat(43); +const NONCE = "B".repeat(43); + +function startupPayload() { + return { + status: "protected", + routingKind: "opencodex-local", + routingInjected: true, + localRoutingDependency: true, + autostartEnabled: true, + rebootSafe: true, + protection: "service", + serviceInstalled: true, + serviceViable: true, + serviceEnabled: true, + serviceRunning: true, + serviceStale: false, + serviceConflict: false, + shimInstalled: true, + shimHealthy: true, + shimCoverage: "cli-only", + serviceSupported: true, + platform: "linux", + diagnosticStale: false, + recommendedCommand: null, + commands: { + installService: "ocx service install", + repairService: "ocx service repair", + installShim: "ocx codex-shim install", + restoreNative: "ocx restore", + }, + }; +} + +function deps(body: unknown) { + return { + readRuntime: () => ({ + pid: LIVE.pid, + port: LIVE.port, + hostname: LIVE.hostname, + attestationSecret: SECRET, + }), + createNonce: () => NONCE, + now: () => 1_000, + fetchImpl: async () => new Response(JSON.stringify(body), { + status: 200, + headers: { "content-type": "application/json" }, + }), + }; +} + +describe("ocx status live startup health", () => { + test("uses an attested live startup verdict when the shell-local service probe would disagree", async () => { + const observed = await fetchLiveStartupHealth(LIVE, deps(startupPayload())); + expect(observed?.status).toBe("protected"); + expect(observed?.rebootSafe).toBe(true); + expect(observed?.serviceViable).toBe(true); + expect(observed?.protection).toBe("service"); + }); + + test("rejects malformed live startup payloads", async () => { + for (const malformed of [ + { ...startupPayload(), serviceRunning: "yes" }, + (() => { const row = { ...startupPayload() } as Record; delete row.platform; return row; })(), + { ...startupPayload(), recommendedCommand: 7 }, + { ...startupPayload(), commands: { installService: "ok" } }, + { ...startupPayload(), routingAdoption: { adoption: "adopted", injectedAtMs: 1, staleClients: "bad", observedClients: 1 } }, + { ...startupPayload(), routingAdoption: { adoption: "adopted", injectedAtMs: 1, staleClients: [{ pid: "bad", startedAtMs: 1 }], observedClients: 1 } }, + ]) { + expect(await fetchLiveStartupHealth(LIVE, deps(malformed))).toBeNull(); + } + }); + + test("selection prefers the attested live verdict and does not evaluate the conflicting fallback", () => { + const live = startupPayload() as StartupHealth; + let fallbackCalls = 0; + const selected = selectStatusStartupHealth(live, () => { + fallbackCalls += 1; + return { ...live, status: "at-risk", rebootSafe: false, protection: "none" } as StartupHealth; + }); + expect(selected.status).toBe("protected"); + expect(selected.rebootSafe).toBe(true); + expect(fallbackCalls).toBe(0); + expect(statusServiceSummary(live, { installed: false, summary: "systemd not found" }, true)) + .toContain("running under the live managed service"); + }); + + test("selection falls back to local startup diagnostics when the live read is unavailable", () => { + const local = { ...startupPayload(), status: "at-risk", rebootSafe: false, protection: "none" } as StartupHealth; + let fallbackCalls = 0; + const selected = selectStatusStartupHealth(null, () => { fallbackCalls += 1; return local; }); + expect(selected).toBe(local); + expect(fallbackCalls).toBe(1); + expect(statusServiceSummary(null, { installed: true, summary: "registered" }, false)) + .toContain("registered but NOT serving"); + }); + + test("service summary never contradicts a present negative live startup verdict", () => { + const live = { + ...startupPayload(), + status: "at-risk", + rebootSafe: false, + protection: "none", + serviceInstalled: false, + serviceRunning: false, + serviceViable: false, + recommendedCommand: "ocx service repair", + } as StartupHealth; + const summary = statusServiceSummary(live, { installed: true, summary: "healthy local service" }, true); + expect(summary).toContain("live startup reports service absent, not running, not viable"); + expect(summary).toContain("ocx service repair"); + expect(summary).not.toContain("healthy local service"); + }); + + test("fails closed when the runtime attestation cannot bind the live PID", async () => { + const observed = await fetchLiveStartupHealth(LIVE, { + ...deps(startupPayload()), + readRuntime: () => null, + }); + expect(observed).toBeNull(); + }); +}); \ No newline at end of file From 6341da9847ea7f03d7a5dd8f3fb9a3f684ab0097 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 17:03:43 +0900 Subject: [PATCH 56/75] fix(status): trust a live startup verdict only with the server proof Follow-up to #5977, closing the hold that kept it out of round 2. The local-read capability authenticates the request, but the client trusted the answer on shape alone, so a process that took the port could supply a protected verdict. The server now signs each local-read response with the attestation proof over the request nonce, PID and port, fetchBoundLocalManagementRead verifies it when a caller opts in, and ocx status opts in. Tests cover the signed server response and unsigned, wrong-nonce and wrong-secret answers. --- scripts/test-layout/layout.json | 2 +- src/cli/status.ts | 2 +- src/server/index/serve-options.ts | 12 ++++ src/server/local-management-read-client.ts | 23 +++++++- structure/gui-and-management-api.md | 7 +++ tests/cli/cli-status-startup-health.test.ts | 17 +++++- tests/fixtures/test-layout-expected.json | 1 + .../server/local-read-response-proof.test.ts | 58 +++++++++++++++++++ 8 files changed, 115 insertions(+), 7 deletions(-) create mode 100644 tests/server/local-read-response-proof.test.ts diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 7fcccbcc734..8b74ac2d7fb 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -173,7 +173,7 @@ "cli-kiro-auto-selection.test.ts": "cli", "codebuddy-live-models.test.ts": "providers", "kiro-auto-selection.test.ts": "providers/kiro", "kiro-quota-metrics.test.ts": "providers/kiro", "management-provider-request-pacing.test.ts": "server", "desktop-supervised-restart.test.ts": "clients", "cli-restart-handoff.test.ts": "cli", "restart-replacement.test.ts": "server", "deepseek-artifact-tool-schema.test.ts": "providers", - "client-config-export-output-limit.test.ts": "config", "responses-skills-snapshot.test.ts": "responses", "responses-code-mode-mcp-direct.test.ts": "responses", + "client-config-export-output-limit.test.ts": "config", "responses-skills-snapshot.test.ts": "responses", "responses-code-mode-mcp-direct.test.ts": "responses", "local-read-response-proof.test.ts": "server", "openai-chat-serialized-tool-call-scaling.test.ts": "adapters/openai", "openai-chat-tool-call-id-remint.test.ts": "adapters/openai", "coding-agent-json-lines-scaling.test.ts": "providers", diff --git a/src/cli/status.ts b/src/cli/status.ts index fd55aaecd4f..d1c5c1fb417 100644 --- a/src/cli/status.ts +++ b/src/cli/status.ts @@ -249,7 +249,7 @@ export async function fetchLiveStartupHealth( deps: Parameters[2] = {}, ): Promise { const result = await fetchBoundLocalManagementRead( - live, LOCAL_MANAGEMENT_READ_PATHS.startupHealth, { timeoutMs: 1_500, ...deps }, + live, LOCAL_MANAGEMENT_READ_PATHS.startupHealth, { timeoutMs: 1_500, ...deps, requireResponseProof: true }, ); if (result.kind !== "response" || !result.response.ok) return null; let payload: unknown; diff --git a/src/server/index/serve-options.ts b/src/server/index/serve-options.ts index 6988a4090a6..8f545330708 100644 --- a/src/server/index/serve-options.ts +++ b/src/server/index/serve-options.ts @@ -170,6 +170,7 @@ import { createLocalAttestationProof, } from "../../lib/local-management-attestation"; import { SYSTEM_RESTART_CAPABILITY_VERSION } from "../../lib/system-restart-contract"; +import { LOCAL_MANAGEMENT_NONCE_HEADER } from "../../lib/local-management-capability"; import { LOCAL_PROVIDER_RELOAD_CAPABILITY_VERSION } from "../../lib/local-provider-reload-contract"; import { LOCAL_ASIDE_SYNC_CAPABILITY_VERSION } from "../../lib/local-aside-sync-contract"; import { @@ -697,6 +698,17 @@ export function createServeOptions(ctx: ServeOptionsContext) { trustedLoopback: trustedLoopbackForIngress(ingress, config.hostname ?? "127.0.0.1"), guiSessionIssuance: managementSessionIssuance(req, managementAuth), }); + // A local read capability authenticates the request; sign its single-use nonce so the + // caller can tell this answer came from this process and not from whoever holds the port. + const readNonce = principal === "local-read-capability" ? req.headers.get(LOCAL_MANAGEMENT_NONCE_HEADER) : null; + const readProof = readNonce + ? createLocalAttestationProof(localAttestationSecret, readNonce, process.pid, localManagementAuth.port) : null; + if (mgmtResponse && readProof) { + const headers = new Headers(mgmtResponse.headers); + headers.set(LOCAL_ATTESTATION_PROOF_HEADER, readProof); + const signed = new Response(mgmtResponse.body, { status: mgmtResponse.status, statusText: mgmtResponse.statusText, headers }); + return withManagementCors(signed, req, config); + } if (mgmtResponse) return withManagementCors(mgmtResponse, req, config); return withManagementCors(formatErrorResponse(404, "not_found", `Unknown endpoint: ${req.method} ${url.pathname}`), req, config); } diff --git a/src/server/local-management-read-client.ts b/src/server/local-management-read-client.ts index e0d64890438..c932d578028 100644 --- a/src/server/local-management-read-client.ts +++ b/src/server/local-management-read-client.ts @@ -1,5 +1,9 @@ import { readRuntimePort, type RuntimePortState } from "../config/process-state"; -import { createLocalAttestationChallenge } from "../lib/local-management-attestation"; +import { + LOCAL_ATTESTATION_PROOF_HEADER, + createLocalAttestationChallenge, + verifyLocalAttestationProof, +} from "../lib/local-management-attestation"; import { LOCAL_MANAGEMENT_CAPABILITY_HEADER, LOCAL_MANAGEMENT_CAPABILITY_EXPIRES_AT_HEADER, @@ -14,7 +18,10 @@ import { probeHostname, type LiveProxy } from "./proxy-liveness"; export type LocalManagementReadResult = | { kind: "response"; response: Response; targetPid: number } - | { kind: "unavailable"; reason: "unattested-target" | "runtime-mismatch" | "capability-unavailable" | "transport" }; + | { + kind: "unavailable"; + reason: "unattested-target" | "runtime-mismatch" | "capability-unavailable" | "transport" | "unattested-response"; + }; export interface LocalManagementReadDeps { fetchImpl?: typeof fetch; @@ -25,6 +32,12 @@ export interface LocalManagementReadDeps { export interface LocalManagementReadRequestDeps extends LocalManagementReadDeps { timeoutMs?: number; + /** + * Require the response to carry the server's attestation over this request's nonce. The + * capability authenticates the request to the real server; only this proof tells the caller + * that the answer came from it and not from whatever process now holds the port. + */ + requireResponseProof?: boolean; } /** @@ -83,6 +96,12 @@ export async function fetchBoundLocalManagementRead( signal: AbortSignal.timeout(deps.timeoutMs ?? 4_000), }, ); + if (deps.requireResponseProof && !verifyLocalAttestationProof( + runtime.attestationSecret, nonce, target.pid, target.port, response.headers.get(LOCAL_ATTESTATION_PROOF_HEADER), + )) { + void response.body?.cancel().catch(() => {}); + return { kind: "unavailable", reason: "unattested-response" }; + } return { kind: "response", response, targetPid: target.pid }; } catch { return { kind: "unavailable", reason: "transport" }; diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index be59cb022dd..b7809b380ab 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -88,6 +88,13 @@ capability, and an unexpected management response so a reachable `401` cannot be their detailed CLI health remains unavailable until restarted with an attested runtime record and capability-aware server. +The server signs every response to a local-read capability with `x-opencodex-attestation-proof` +over that request's nonce, PID, and port. A caller that decides something from the answer opts in +with `requireResponseProof`, which refuses an unsigned or mismatched response: the capability +proves the request to the server, and only the proof tells the caller that the answer came from it +and not from a process that took the port. `ocx status` requires the proof before it prefers the +live proxy's `/api/startup-health` verdict over its own shell-local probe (#5977). + The desktop tray uses the same read-v1 contract for its fixed GET allowlist in `src/lib/local-management-capability.ts`, including `/api/oauth/accounts` and `/api/providers/keys`. Provider selectors and `quota=1` remain signed into the diff --git a/tests/cli/cli-status-startup-health.test.ts b/tests/cli/cli-status-startup-health.test.ts index fd51366b7ad..80eadb1fa08 100644 --- a/tests/cli/cli-status-startup-health.test.ts +++ b/tests/cli/cli-status-startup-health.test.ts @@ -1,6 +1,7 @@ import { describe, expect, test } from "bun:test"; import { fetchLiveStartupHealth, selectStatusStartupHealth, statusServiceSummary } from "../../src/cli/status"; import type { StartupHealth } from "../../src/codex/autostart-health"; +import { LOCAL_ATTESTATION_PROOF_HEADER, createLocalAttestationProof } from "../../src/lib/local-management-attestation"; const LIVE = { pid: 4242, @@ -43,7 +44,7 @@ function startupPayload() { }; } -function deps(body: unknown) { +function deps(body: unknown, proof: string | null = createLocalAttestationProof(SECRET, NONCE, LIVE.pid, LIVE.port)) { return { readRuntime: () => ({ pid: LIVE.pid, @@ -55,7 +56,7 @@ function deps(body: unknown) { now: () => 1_000, fetchImpl: async () => new Response(JSON.stringify(body), { status: 200, - headers: { "content-type": "application/json" }, + headers: { "content-type": "application/json", ...(proof ? { [LOCAL_ATTESTATION_PROOF_HEADER]: proof } : {}) }, }), }; } @@ -123,6 +124,16 @@ describe("ocx status live startup health", () => { expect(summary).not.toContain("healthy local service"); }); + // Whoever holds the port can answer the request; only the server that owns the runtime secret + // can sign this request's nonce. A substituted listener's verdict must never be trusted. + test("rejects a live verdict without the server's proof over this request's nonce", async () => { + expect(await fetchLiveStartupHealth(LIVE, deps(startupPayload(), null))).toBeNull(); + const otherNonce = createLocalAttestationProof(SECRET, "C".repeat(43), LIVE.pid, LIVE.port); + expect(await fetchLiveStartupHealth(LIVE, deps(startupPayload(), otherNonce))).toBeNull(); + const otherSecret = createLocalAttestationProof("D".repeat(43), NONCE, LIVE.pid, LIVE.port); + expect(await fetchLiveStartupHealth(LIVE, deps(startupPayload(), otherSecret))).toBeNull(); + }); + test("fails closed when the runtime attestation cannot bind the live PID", async () => { const observed = await fetchLiveStartupHealth(LIVE, { ...deps(startupPayload()), @@ -130,4 +141,4 @@ describe("ocx status live startup health", () => { }); expect(observed).toBeNull(); }); -}); \ No newline at end of file +}); diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index f53b8b96c4a..bfffa0d289d 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -1394,6 +1394,7 @@ "responses-canonical-only-top-level-fields.test.ts": "responses", "responses-code-mode-goal-helpers.test.ts": "responses", "responses-code-mode-mcp-direct.test.ts": "responses", + "local-read-response-proof.test.ts": "server", "responses-code-mode-patch-compile.test.ts": "responses", "responses-code-mode-shell-compile.test.ts": "responses", "responses-compact-handoff-admission.test.ts": "responses", diff --git a/tests/server/local-read-response-proof.test.ts b/tests/server/local-read-response-proof.test.ts new file mode 100644 index 00000000000..c444f6d6f6b --- /dev/null +++ b/tests/server/local-read-response-proof.test.ts @@ -0,0 +1,58 @@ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { startServer } from "../../src/server"; +import { + LOCAL_ATTESTATION_PROOF_HEADER, + verifyLocalAttestationProof, +} from "../../src/lib/local-management-attestation"; +import { + LOCAL_MANAGEMENT_CAPABILITY_EXPIRES_AT_HEADER, + LOCAL_MANAGEMENT_CAPABILITY_HEADER, + LOCAL_MANAGEMENT_CAPABILITY_TTL_MS, + LOCAL_MANAGEMENT_EXPECTED_PID_HEADER, + LOCAL_MANAGEMENT_NONCE_HEADER, + LOCAL_MANAGEMENT_READ_PATHS, + createLocalManagementReadCapability, +} from "../../src/lib/local-management-capability"; +import { createTempHome, type TempHome } from "../helpers/temp-home"; + +/** + * A local read capability authenticates the request to the server. The answer is only worth + * trusting (for example as `ocx status`'s startup verdict, #5977) when it carries the server's + * attestation over the same single-use nonce. + */ +let home: TempHome; +beforeEach(() => { home = createTempHome("ocx-local-read-proof-"); }); +afterEach(() => { home.remove(); }); + +test("a local-read response carries the server's proof over the request nonce", async () => { + const secret = "A".repeat(43); + const nonce = "B".repeat(43); + const server = startServer(0, { + localAttestationSecret: secret, + managementAuthState: { available: false, reason: "injected unavailable state" }, + }); + try { + const path = LOCAL_MANAGEMENT_READ_PATHS.systemMemory; + const expiresAt = Date.now() + LOCAL_MANAGEMENT_CAPABILITY_TTL_MS; + const response = await fetch(new URL(path, server.url), { headers: { + [LOCAL_MANAGEMENT_EXPECTED_PID_HEADER]: String(process.pid), + [LOCAL_MANAGEMENT_NONCE_HEADER]: nonce, + [LOCAL_MANAGEMENT_CAPABILITY_EXPIRES_AT_HEADER]: String(expiresAt), + [LOCAL_MANAGEMENT_CAPABILITY_HEADER]: createLocalManagementReadCapability( + secret, nonce, "GET", path, process.pid, server.port!, expiresAt, + )!, + } }); + expect(response.status).toBe(200); + const proof = response.headers.get(LOCAL_ATTESTATION_PROOF_HEADER); + expect(verifyLocalAttestationProof(secret, nonce, process.pid, server.port!, proof)).toBe(true); + expect(verifyLocalAttestationProof(secret, "C".repeat(43), process.pid, server.port!, proof)).toBe(false); + await response.text(); + + // Without the capability there is no read, so nothing is signed. + const refused = await fetch(new URL(path, server.url)); + expect(refused.headers.get(LOCAL_ATTESTATION_PROOF_HEADER)).toBeNull(); + await refused.text(); + } finally { + await server.stop(true); + } +}); From 58395b52a98b73d7237f7b8e9c34e2b4546adb7f Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 17:10:08 +0900 Subject: [PATCH 57/75] docs(devlog): record train 3 B6 evidence --- .../_plan/260927_merge_train_3/060_batch6.md | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/devlog/_plan/260927_merge_train_3/060_batch6.md b/devlog/_plan/260927_merge_train_3/060_batch6.md index fabff72f5f0..d61b6257931 100644 --- a/devlog/_plan/260927_merge_train_3/060_batch6.md +++ b/devlog/_plan/260927_merge_train_3/060_batch6.md @@ -13,3 +13,22 @@ Held from this round's reviews, with reasons, for the outcome ledger: #4177 (no feature), #4732 (perf rework against `snapshot-select.ts` and measurements needed from the author), #5539 (reverses test-locked behavior without a reproduction), #4961 (issue withholds a design), #4143 (needs the reporter's desktop routing details). + +## Build and evidence + +| Commit | What | +|---|---| +| `bee2ea4357` | #5925 carried (four commits squashed, author kept) | +| `2a383cbc1f` | its layout entry moved onto a shared line (layout.json stays at 1993 lines) | +| `f2727befa8` | #5977 carried; the author's noreply identity replaces the placeholder address on the commit | +| `6341da9847` | #5977 response proof: the server signs local-read responses over the request nonce, the client verifies when asked, `ocx status` asks. New `tests/server/local-read-response-proof.test.ts` and a negative client test; both fail without the change | + +Security: #5925 dedicated review BLOCKER no. #5977's remaining hold is closed by `6341da9847`, which reuses the +attestation that `/healthz` and system restart already use. + +Aside: #5925 shows no open review; #5977 shows two approvals from before the response-proof commit. + +Local proof at `6341da9847`: typecheck, structure and privacy exit 0; the seven local-read and status files plus the +three #5925 files 122 pass. Directory runs: `tests/adapters` 2371 pass, 107 fail on this branch and 2369 pass, 107 fail +on `dev` `7d8459388c` (same Anthropic cooldown and pool files, which pass alone), so the failures are pre-existing +directory-run interference; `tests/responses` shows the 13 known `responses-compaction-recovery` failures. From b18129627e767f7948c8db7adb3d2be76457f8e6 Mon Sep 17 00:00:00 2001 From: JUN Date: Sun, 27 Sep 2026 17:32:41 +0900 Subject: [PATCH 58/75] docs(devlog): plan train 3 B7 --- devlog/_plan/260927_merge_train_3/070_batch7.md | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) create mode 100644 devlog/_plan/260927_merge_train_3/070_batch7.md diff --git a/devlog/_plan/260927_merge_train_3/070_batch7.md b/devlog/_plan/260927_merge_train_3/070_batch7.md new file mode 100644 index 00000000000..84b82f24e6e --- /dev/null +++ b/devlog/_plan/260927_merge_train_3/070_batch7.md @@ -0,0 +1,16 @@ +# B7 — GUI bug fixes + +Base: `dev` `429f4e0175` (after B6 #6069). Branch `codex/train3-b7`. + +Previous D (B6): #5925 and #5977 landed. Scope note: the request covers bugs, and only enhancements must avoid the +GUI, so GUI bug fixes are in scope; earlier batches skipped them by a stricter reading. + +| PR | Author | Change | Kimi | UI visible | +|---|---|---|---|---| +| #6025 | Ingwannu | Kiro device-login status reads get one bounded, cancellable operation (fetch, body, decode), so a stalled body can no longer hang the dialog or the finalizer | LAND; its test fails on dev | no | +| #6010 | Ingwannu | The provider deep-link test stops dispatching a second `hashchange` for a changed hash | LAND; flake from CI, not reproduced locally | no (test only) | +| #6007 | Ingwannu | With provider-table routing, the dashboard and the start/sync output warn that some mobile remote thread lists hide openai-tagged history (#5848 mitigation; the issue stays open) | APPROVE; two dev tests fail without it | yes: a hint under the authless or client-compaction switch | + +The batch PR needs a screenshot for #6007. It is taken from this branch's proxy run with `HOME`, `OPENCODEX_HOME` +and `CODEX_HOME` all pointed at a temporary directory, so no real shell profile, Codex config or app integration is +touched, and uploaded through the `pr-assets` branch. From 960e482b9e4f1e1aa35b82272d6d661e90dc9471 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 17:32:44 +0900 Subject: [PATCH 59/75] fix(gui): bound Kiro status reconciliation (#6025) Carried from #6025 into merge train round 3. Co-authored-by: Ingwannu --- .../100_gui_device_login_and_skip_reason.md | 6 +- .../src/content/docs/guides/providers.md | 2 +- gui/src/components/use-kiro-device-login.ts | 41 ++++--- gui/src/kiro-device-login-finalizer.ts | 104 ++++++++++++++---- gui/tests/kiro-device-login.test.tsx | 90 +++++++++++++-- .../ADR-6021-kiro-status-read-ownership.md | 12 ++ structure/gui-and-management-api.md | 4 +- 7 files changed, 213 insertions(+), 46 deletions(-) create mode 100644 structure/decisions/ADR-6021-kiro-status-read-ownership.md diff --git a/devlog/_plan/260926_kiro_lb_parity2/100_gui_device_login_and_skip_reason.md b/devlog/_plan/260926_kiro_lb_parity2/100_gui_device_login_and_skip_reason.md index 7b6b968df0d..6f7179c8d89 100644 --- a/devlog/_plan/260926_kiro_lb_parity2/100_gui_device_login_and_skip_reason.md +++ b/devlog/_plan/260926_kiro_lb_parity2/100_gui_device_login_and_skip_reason.md @@ -215,7 +215,11 @@ cd docs-site && bun run build as a server follow-up candidate in 000 and in the PR. - **A8 In-flight terminal replies win.** The hook (and finalizer) processes a terminal reply from a status request already in flight even after Cancel; a later 404 still takes the neutral path (A3). A1's promise - becomes: success is shown only after a terminal `done` status reply is observed. + becomes: success is shown only after a terminal `done` status reply is observed. The handoff owns one + parsed status-read operation rather than cloned `Response` bodies. Its 45 s budget covers fetch, body EOF + and JSON parsing; the finalizer also cancels that reader when the flow-wide deadline wins and never waits + a full retry interval beyond the deadline. This preserves the already-sent terminal reply without letting + a stalled body retain the module-scoped singleflight entry indefinitely (#6021). - **A9 Settle split at the guard.** Two helpers: `reloadAccountsAfterLogin(provider)` (awaited `fetchAccountSets`) and `refreshDerivedAfterLogin()` (`fetchConfig`, `fetchProviderQuotas(true)`, `bumpModelsRefresh`). The existing loop keeps its generation/mounted guard **between** them, keeps its diff --git a/docs-site/src/content/docs/guides/providers.md b/docs-site/src/content/docs/guides/providers.md index 6a91e7fe200..b8f30a1ca97 100644 --- a/docs-site/src/content/docs/guides/providers.md +++ b/docs-site/src/content/docs/guides/providers.md @@ -451,7 +451,7 @@ an ambiguous token selection), when `KIROCLI_DB_PATH` / `KIRO_CLI_DB_FILE` redir from the live CLI store, or when an existing primary CLI database has no recognized token row. Repair or remove the unreadable database under the normal `kiro-cli` data path, unset those import selectors, then retry. Signing in from a machine with no existing `kiro-cli` session is unaffected. -The native dashboard choices are add-only and do not sign out `kiro-cli`. A device dialog shows the code and verification destination. Only recognized Kiro or Builder ID hosts are opened as links; an unexpected destination is shown as copyable text for review. +The native dashboard choices are add-only and do not sign out `kiro-cli`. A device dialog shows the code and verification destination. Only recognized Kiro or Builder ID hosts are opened as links; an unexpected destination is shown as copyable text for review. Closing the dialog sends cancellation and leaves a bounded background status check to reconcile a commit already in progress. A stalled status response is retried; exhausting the flow deadline produces the neutral ended outcome rather than claiming success. The account list marks Kiro accounts excluded from automatic selection with a reason, when available. ## 3. API-key catalog diff --git a/gui/src/components/use-kiro-device-login.ts b/gui/src/components/use-kiro-device-login.ts index 1e8d2c92379..e4c7e978d39 100644 --- a/gui/src/components/use-kiro-device-login.ts +++ b/gui/src/components/use-kiro-device-login.ts @@ -1,12 +1,15 @@ import { useCallback, useEffect, useRef, useState } from "react"; import { afterOAuthCancellation } from "../oauth-cancellation-barrier"; -import { finalizeKiroDeviceFlow, observeKiroDeviceFinal, type KiroFinalOutcome } from "../kiro-device-login-finalizer"; +import { + finalizeKiroDeviceFlow, observeKiroDeviceFinal, readKiroDeviceStatus, + type KiroFinalOutcome, type KiroStatusRead, +} from "../kiro-device-login-finalizer"; import { parseKiroDeviceView, type KiroDeviceMethod, type KiroDeviceView } from "../kiro-device-login-helpers"; type Phase = "idle" | "starting" | "pending" | "done" | "expired" | "failed" | "cancelled" | "ended"; export type KiroLoginState = { phase: Phase; view?: KiroDeviceView; error?: "start" | "network" | "invalid" }; -type Session = { closed: boolean; view?: KiroDeviceView; inFlight?: Promise; - waitController?: AbortController; terminal?: KiroFinalOutcome }; +type Session = { closed: boolean; closedController: AbortController; view?: KiroDeviceView; + inFlight?: KiroStatusRead; waitController?: AbortController; terminal?: KiroFinalOutcome }; const wait = (ms: number, signal: AbortSignal) => new Promise(resolve => { if (signal.aborted) { resolve(); return; } const timer = setTimeout(() => { signal.removeEventListener("abort", stop); resolve(); }, ms); @@ -16,7 +19,19 @@ const wait = (ms: number, signal: AbortSignal) => new Promise(resolve => { const CLOSED = Symbol("closed"); /** The awaited value, or CLOSED when the session closed while it was pending (a late reply belongs to the finalizer). */ const unlessClosed = (session: Session, value: Promise): Promise => - value.then(result => (session.closed ? CLOSED : result)); + new Promise(resolve => { + if (session.closed) { resolve(CLOSED); return; } + let done = false; + const settle = (result: T | typeof CLOSED) => { + if (done) return; + done = true; + session.closedController.signal.removeEventListener("abort", stop); + resolve(result); + }; + const stop = () => settle(CLOSED); + session.closedController.signal.addEventListener("abort", stop, { once: true }); + void value.then(result => settle(session.closed ? CLOSED : result), () => settle(CLOSED)); + }); export function useKiroDeviceLogin(apiBase: string, onSettled?: (provider: string, outcome: KiroFinalOutcome) => void, pollDelay: (ms: number, signal: AbortSignal) => Promise = wait) { @@ -44,6 +59,7 @@ export function useKiroDeviceLogin(apiBase: string, onSettled?: (provider: strin const session = sessionRef.current; if (!session || session.closed) return; session.closed = true; + session.closedController.abort(); sessionRef.current = null; session.waitController?.abort(); if (mountedRef.current) setState({ phase: "cancelled" }); @@ -64,7 +80,7 @@ export function useKiroDeviceLogin(apiBase: string, onSettled?: (provider: strin const start = useCallback(async (method: KiroDeviceMethod) => { if (sessionRef.current) return; - const session: Session = { closed: false }; + const session: Session = { closed: false, closedController: new AbortController() }; sessionRef.current = session; setState({ phase: "starting" }); let response: Response | undefined; @@ -114,21 +130,18 @@ export function useKiroDeviceLogin(apiBase: string, onSettled?: (provider: strin break; } const flowId = view.flowId; - const request = fetch(`${apiBase}/api/oauth/status?provider=kiro&flowId=${encodeURIComponent(flowId)}`).catch(() => null); - session.inFlight = request.then(response => response?.clone() ?? null); - const status = await unlessClosed(session, request); + const request = readKiroDeviceStatus(apiBase, flowId); + session.inFlight = request; + const status = await unlessClosed(session, request.result); if (status === CLOSED) break; - if (status?.status === 404) { - session.inFlight = undefined; + session.inFlight = undefined; + if (status.kind === "missing") { session.terminal = "ended"; settledRef.current?.("kiro", "ended"); setState({ phase: "ended", view: session.view }); break; } - const body = status?.ok ? await unlessClosed(session, status.json().catch(() => null)) : null; - if (body === CLOSED) break; - const next = body === null ? null : parseKiroDeviceView(body); - session.inFlight = undefined; + const next = status.kind === "view" ? status.view : null; if (!next || next.flowId !== flowId) continue; session.view = next; if (next.state === "pending") { setState({ phase: "pending", view: next }); continue; } diff --git a/gui/src/kiro-device-login-finalizer.ts b/gui/src/kiro-device-login-finalizer.ts index a2327ea397f..dff5ad25890 100644 --- a/gui/src/kiro-device-login-finalizer.ts +++ b/gui/src/kiro-device-login-finalizer.ts @@ -1,16 +1,84 @@ import { parseKiroDeviceView, type KiroDeviceView } from "./kiro-device-login-helpers"; export type KiroFinalOutcome = "added" | "ended" | "failed"; +export type KiroStatusResult = + | { kind: "view"; view: KiroDeviceView } + | { kind: "missing" } + | { kind: "retry" }; +export type KiroStatusRead = { result: Promise; cancel: () => void }; type Listener = (outcome: KiroFinalOutcome) => void; const active = new Map>(); const terminal = new Map(); const listeners = new Map>(); const keyFor = (apiBase: string, flowId: string) => JSON.stringify([apiBase, flowId]); const delay = (ms: number) => new Promise(resolve => setTimeout(resolve, ms)); -async function boundedRead(read: Promise, ms: number): Promise { +const MAX_STATUS_BODY_BYTES = 64 * 1024; + +function cancelBody(response: Response): void { + try { void response.body?.cancel().catch(() => {}); } catch { /* best effort */ } +} + +/** Own fetch, body EOF and parsing as one cancellable operation; headers alone are not completion. */ +export function readKiroDeviceStatus(apiBase: string, flowId: string, timeoutMs = 45_000): KiroStatusRead { + const controller = new AbortController(); + let reader: ReadableStreamDefaultReader | undefined; + let settled = false; + let finish!: (result: KiroStatusResult) => void; + const result = new Promise(resolve => { finish = resolve; }); + const settle = (value: KiroStatusResult) => { + if (settled) return; + settled = true; + clearTimeout(timer); + finish(value); + }; + const cancel = () => { + if (settled) return; + controller.abort(); + try { void reader?.cancel().catch(() => {}); } catch { /* best effort */ } + // Transport abort and stream cancellation are cooperative. The caller's deadline is not. + settle({ kind: "retry" }); + }; + const timer = setTimeout(cancel, Math.max(0, timeoutMs)); + void (async () => { + try { + const response = await fetch( + `${apiBase}/api/oauth/status?provider=kiro&flowId=${encodeURIComponent(flowId)}`, + { signal: controller.signal }, + ); + if (settled) { cancelBody(response); return; } + if (response.status === 404) { cancelBody(response); settle({ kind: "missing" }); return; } + if (!response.ok || !response.body) { cancelBody(response); settle({ kind: "retry" }); return; } + reader = response.body.getReader(); + const decoder = new TextDecoder(); + let text = ""; + let bytes = 0; + while (true) { + const chunk = await reader.read(); + if (settled) return; + if (chunk.done) break; + bytes += chunk.value.byteLength; + if (bytes > MAX_STATUS_BODY_BYTES) { cancel(); return; } + text += decoder.decode(chunk.value, { stream: true }); + } + text += decoder.decode(); + let decoded: unknown; + try { decoded = JSON.parse(text); } catch { settle({ kind: "retry" }); return; } + const view = parseKiroDeviceView(decoded); + settle(view ? { kind: "view", view } : { kind: "retry" }); + } catch { settle({ kind: "retry" }); } + })(); + return { result, cancel }; +} + +async function awaitStatusRead(read: KiroStatusRead, ms: number): Promise { let timer: ReturnType | undefined; try { - return await Promise.race([read, new Promise(resolve => { timer = setTimeout(() => resolve(null), ms); })]); + return await Promise.race([ + read.result, + new Promise(resolve => { + timer = setTimeout(() => { read.cancel(); resolve({ kind: "retry" }); }, Math.max(0, ms)); + }), + ]); } finally { if (timer) clearTimeout(timer); } } @@ -45,7 +113,7 @@ export function observeKiroDeviceFinal(apiBase: string, flowId: string, view: Ki /** Detached reconciliation survives dialog and page unmount. One loop per flowId. */ export function finalizeKiroDeviceFlow(apiBase: string, flowId: string, expiresAt?: number, - inFlight?: Promise): Promise { + inFlight?: KiroStatusRead): Promise { const key = keyFor(apiBase, flowId); const prior = active.get(key); if (prior) return prior; @@ -55,27 +123,23 @@ export function finalizeKiroDeviceFlow(apiBase: string, flowId: string, expiresA const run = (async () => { let pending = inFlight; while (Date.now() < deadline) { - let response: Response | null; - try { - const remaining = deadline - Date.now(); - const timeout = Math.min(45_000, remaining); - response = await boundedRead(pending ?? fetch( - `${apiBase}/api/oauth/status?provider=kiro&flowId=${encodeURIComponent(flowId)}`, - { signal: AbortSignal.timeout(timeout) }, - ), timeout); - } catch { response = null; } + const remaining = deadline - Date.now(); + const read = pending ?? readKiroDeviceStatus(apiBase, flowId, Math.min(45_000, remaining)); pending = undefined; + const status = await awaitStatusRead(read, remaining); if (terminal.has(key)) return terminal.get(key)!; - if (response?.status === 404) return finish(apiBase, flowId, "ended"); - if (response?.ok) { - const view = parseKiroDeviceView(await response.json().catch(() => null)); - if (view) { - const result = observeKiroDeviceFinal(apiBase, flowId, view); - if (result) return result; - } + if (status.kind === "missing") return finish(apiBase, flowId, "ended"); + if (status.kind === "view") { + const result = observeKiroDeviceFinal(apiBase, flowId, status.view); + if (result) return result; } - await delay(2_000); + const retryRemaining = deadline - Date.now(); + if (retryRemaining <= 0) break; + await delay(Math.min(2_000, retryRemaining)); } + // A close can hand off after expiry. Do not leave that inherited transport alive merely + // because there was no remaining loop iteration in which the deadline wrapper could cancel it. + pending?.cancel(); return finish(apiBase, flowId, "ended"); })().finally(() => { active.delete(key); }); active.set(key, run); diff --git a/gui/tests/kiro-device-login.test.tsx b/gui/tests/kiro-device-login.test.tsx index 8be8c8326e3..203b1d5ee6c 100644 --- a/gui/tests/kiro-device-login.test.tsx +++ b/gui/tests/kiro-device-login.test.tsx @@ -11,8 +11,9 @@ import type { ProviderAuthHandlers } from "../src/components/provider-workspace/ import { en } from "../src/i18n/en"; import { interpolate, type TFn } from "../src/i18n/shared"; import { useProvidersOAuth } from "../src/pages/use-providers-oauth"; -import { finalizeKiroDeviceFlow } from "../src/kiro-device-login-finalizer"; -import { subscribeKiroDeviceFinal } from "../src/kiro-device-login-finalizer"; +import { + finalizeKiroDeviceFlow, readKiroDeviceStatus, subscribeKiroDeviceFinal, +} from "../src/kiro-device-login-finalizer"; const globals = ["document", "window", "navigator", "localStorage", "fetch", "IS_REACT_ACT_ENVIRONMENT"] as const; let previous: Record<(typeof globals)[number], unknown>; @@ -334,14 +335,19 @@ test("close during start dispatches cancel before detached status", async () => unsubscribe(); }); -test("close during status body parsing leaves finalizer as sole outcome owner", async () => { - const body = deferred(); +test("close during a delayed status body transfers its sole reader to the finalizer", async () => { + let bodyController!: ReadableStreamDefaultController; let readingBody = false; responder = async url => { if (url.includes("/api/oauth/status?")) { - const response = json(view("done")); - Object.defineProperty(response, "json", { value: () => { readingBody = true; return body.promise; } }); - return response; + return new Response(new ReadableStream({ + start(controller) { + bodyController = controller; + readingBody = true; + controller.enqueue(new TextEncoder().encode(JSON.stringify(view("done")))); + // The terminal JSON is not complete until EOF; close it only after the component unmounts. + }, + }), { headers: { "Content-Type": "application/json" } }); } return url.endsWith("/api/oauth/login/cancel") ? json(view("cancelled")) : json(view("pending")); }; @@ -352,13 +358,79 @@ test("close during status body parsing leaves finalizer as sole outcome owner", expect(readingBody).toBe(true); await act(async () => { root!.unmount(); root = null; }); await flush(); - expect(received).toEqual(["added"]); - await act(async () => { body.resolve(view("done")); }); await flush(); + expect(received).toEqual([]); + await act(async () => { bodyController.close(); }); await flush(); expect(settled).toEqual([]); expect(received).toEqual(["added"]); + expect(requests.filter(r => r.url.includes("/api/oauth/status?"))).toHaveLength(1); unsubscribe(); }); +test("status read deadline covers a body that never reaches EOF", async () => { + let bodyCancelled = 0; + responder = async url => url.includes("/api/oauth/status?") + ? new Response(new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode(JSON.stringify(view("done")))); + }, + cancel() { bodyCancelled++; }, + }), { headers: { "Content-Type": "application/json" } }) + : json({}); + const read = readKiroDeviceStatus("", flowId, 20); + expect(await read.result).toEqual({ kind: "retry" }); + expect(bodyCancelled).toBe(1); +}); + +test("status read deadline settles when fetch ignores abort and discards its late body", async () => { + const pending = deferred(); + let lateBodyCancelled = 0; + responder = async url => url.includes("/api/oauth/status?") ? pending.promise : json({}); + const read = readKiroDeviceStatus("", flowId, 20); + expect(await read.result).toEqual({ kind: "retry" }); + pending.resolve(new Response(new ReadableStream({ + cancel() { lateBodyCancelled++; }, + }))); + await Promise.resolve(); + await Promise.resolve(); + await new Promise(resolve => setTimeout(resolve, 0)); + expect(lateBodyCancelled).toBe(1); +}); + +test("the overall finalizer deadline cancels a longer inherited body read", async () => { + let bodyCancelled = 0; + responder = async url => url.includes("/api/oauth/status?") + ? new Response(new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode(JSON.stringify(view("done")))); + }, + cancel() { bodyCancelled++; }, + }), { headers: { "Content-Type": "application/json" } }) + : json({}); + const inherited = readKiroDeviceStatus("", flowId, 45_000); + const first = finalizeKiroDeviceFlow("", flowId, Date.now() - 59_980, inherited); + expect(await first).toBe("ended"); + expect(bodyCancelled).toBe(1); + const afterCleanup = finalizeKiroDeviceFlow("", flowId, Date.now() + 60_000); + expect(afterCleanup).not.toBe(first); + expect(await afterCleanup).toBe("ended"); +}); + +test("an already-expired finalizer cancels its inherited read without polling", async () => { + let bodyCancelled = 0; + responder = async url => url.includes("/api/oauth/status?") + ? new Response(new ReadableStream({ + cancel() { bodyCancelled++; }, + })) + : json({}); + const inherited = readKiroDeviceStatus("", flowId, 45_000); + expect(await finalizeKiroDeviceFlow("", flowId, Date.now() - 60_001, inherited)).toBe("ended"); + await Promise.resolve(); + await Promise.resolve(); + await new Promise(resolve => setTimeout(resolve, 0)); + expect(bodyCancelled).toBe(1); + expect(requests.filter(r => r.url.includes("/api/oauth/status?"))).toHaveLength(1); +}); + test("closing aborts the pending timer and never dispatches a hook status fetch", async () => { let waitSignal: AbortSignal | undefined; const cancellableDelay = (_ms: number, signal: AbortSignal) => new Promise(resolve => { diff --git a/structure/decisions/ADR-6021-kiro-status-read-ownership.md b/structure/decisions/ADR-6021-kiro-status-read-ownership.md new file mode 100644 index 00000000000..599e7b9acf4 --- /dev/null +++ b/structure/decisions/ADR-6021-kiro-status-read-ownership.md @@ -0,0 +1,12 @@ +# ADR-6021 — Kiro status-read ownership + +- Contract owner: [GUI and management API](../gui-and-management-api.md#authentication-boundaries) + +## Decision record + +- Purpose and intent: Let Kiro device-login reconciliation survive dialog unmount without allowing a stalled response body to outlive the documented read and flow deadlines. +- Existing implementation and constraints: The hook consumed an original `Response` while the detached finalizer consumed a clone. The 45-second race ended when headers arrived, so either body could then wait forever. The already-sent status request must still be able to prove a terminal credential commit after Cancel; simply aborting it on unmount would lose that evidence and permit a later 404 to win. +- Alternatives considered: Abort the hook request and start a fresh finalizer request; retain cloned responses but race each `json()` call; transfer one operation that owns transport, body, parsing and cancellation. +- Chosen approach: Create one status-read operation before dispatch. It owns fetch, a bounded 64 KiB body reader, JSON decoding and public-view validation under one 45-second timer. Closing the dialog transfers that operation to the module-scoped finalizer. The finalizer separately caps its wait by the remaining flow deadline, cancels an inherited reader when that bound wins, and clamps retry sleep to the remaining time. +- Why this approach: One reader preserves terminal-reply precedence without duplicate body consumers. Explicit stream ownership lets timeout settle independently of cooperative transport abort and releases the finalizer singleflight even when EOF never arrives. +- Benefits, costs and impact: Status reads have deterministic memory and lifetime bounds, and a delayed terminal EOF can still reconcile after unmount. The helper is intentionally scoped to status polling; initial-login and cancellation response bodies retain their existing behavior. A status body larger than 64 KiB is treated as retryable invalid input rather than retained. diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index b7809b380ab..56ea1763b01 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -58,11 +58,13 @@ unsupported; deployments that previously relied on such embedding must open it a Kiro management login starts the native device flow only when `POST /api/oauth/login` supplies `method: "builder-id"`, `"google"`, or `"github"`. A method-less request retains -the Kiro CLI flow used by the dashboard chooser. The KiroDeviceLoginDialog and useKiroDeviceLogin GUI modules own the native chooser and polling. The kiro-device-login-finalizer GUI module continues terminal status reads after dialog unmount; a provisional cancel result is never treated as confirmed success. Native status and cancellation require `flowId`; +the Kiro CLI flow used by the dashboard chooser. The KiroDeviceLoginDialog and useKiroDeviceLogin GUI modules own the native chooser and polling. The kiro-device-login-finalizer GUI module continues terminal status reads after dialog unmount; a provisional cancel result is never treated as confirmed success. Hook and finalizer share one status-read operation that owns fetch, bounded body consumption, parsing and cancellation. The complete read has a 45-second ceiling, the detached loop remains bounded by flow expiry plus 60 seconds (and 16 minutes maximum), and closing the dialog transfers rather than clones an in-flight reader so an already-observed terminal reply can still win. Native status and cancellation require `flowId`; provider-keyed status and cancellation retain their previous behavior. Device-flow responses contain only the flow handle, method, public verification fields, state, expiry, and a duplicate-profile warning when applicable. The dashboard renders a native device dialog and links only exact Builder ID or Kiro-owned verification hosts. Kiro account rows show automatic-selection exclusion reasons when present. +> Decision record: [ADR-6021](decisions/ADR-6021-kiro-status-read-ownership.md) + OpenCodex uses three mutually exclusive reusable admission credential classes: | Credential class | Sources | Allowed surface | From 519b9d7676c3bb692fa48a46f2ac31253c6bb293 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 17:32:45 +0900 Subject: [PATCH 60/75] test(gui): avoid duplicate provider hash events (#6010) Carried from #6010 into merge train round 3. Co-authored-by: Ingwannu --- gui/tests/providers-deep-link.test.tsx | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/gui/tests/providers-deep-link.test.tsx b/gui/tests/providers-deep-link.test.tsx index 9ee1a6d924b..feac8f80133 100644 --- a/gui/tests/providers-deep-link.test.tsx +++ b/gui/tests/providers-deep-link.test.tsx @@ -59,8 +59,15 @@ async function mount(names: string[] | null, onAccounts?: (name: string, choose: async function hash(next: string) { await act(async () => { + const previous = testWindow.location.hash; testWindow.location.hash = next; - testWindow.dispatchEvent(new testWindow.HashChangeEvent("hashchange")); + // A changed location emits its own hashchange in browsers and Happy DOM. Dispatching + // another one raced that native task and occasionally applied one navigation twice. + // Only an unchanged hash needs the explicit re-application event used by production. + if (testWindow.location.hash === previous) { + testWindow.dispatchEvent(new testWindow.HashChangeEvent("hashchange")); + } + await new Promise(resolve => setTimeout(resolve, 0)); }); } From b51e20ceb06f2f3a9ad4bd8a3bf2842e740fb906 Mon Sep 17 00:00:00 2001 From: Ingwannu Date: Sun, 27 Sep 2026 17:32:47 +0900 Subject: [PATCH 61/75] fix(codex): surface remote provider-history filtering (#6007) Carried from #6007 into merge train round 3. Co-authored-by: Ingwannu --- .../content/docs/guides/codex-integration.md | 21 +++++++++++++++++++ gui/src/i18n/de.ts | 1 + gui/src/i18n/en.ts | 1 + gui/src/i18n/fr.ts | 1 + gui/src/i18n/ja.ts | 1 + gui/src/i18n/ko.ts | 1 + gui/src/i18n/ru.ts | 1 + gui/src/i18n/tr.ts | 1 + gui/src/i18n/vi.ts | 1 + gui/src/i18n/zh-TW.ts | 1 + gui/src/i18n/zh.ts | 1 + gui/src/pages/dashboard-overview-sections.tsx | 4 ++++ gui/tests/vision-sidecar-dashboard.test.tsx | 10 +++++++++ src/codex/inject.ts | 4 ++++ src/codex/inject/routing-target.ts | 7 ++++++- structure/codex-home.md | 4 ++++ ...rovider-table-remote-history-visibility.md | 12 +++++++++++ .../codex-inject-integration.test.ts | 9 ++++++++ tests/codex-integration/codex-inject.test.ts | 21 +++++++++++++++++++ 19 files changed, 101 insertions(+), 1 deletion(-) create mode 100644 structure/decisions/ADR-5848-provider-table-remote-history-visibility.md diff --git a/docs-site/src/content/docs/guides/codex-integration.md b/docs-site/src/content/docs/guides/codex-integration.md index 2089cd136e9..2a95f14de68 100644 --- a/docs-site/src/content/docs/guides/codex-integration.md +++ b/docs-site/src/content/docs/guides/codex-integration.md @@ -1056,6 +1056,27 @@ When returning to the root-override form, OpenCodex retains an existing `[model_ Enabling the integration in its provider-table form on a home whose `openai`-tagged conversations Codex has already paginated used to be refused outright with `history_paginated_openai_requires_native_writer`: nothing was written and the integration stayed disabled. OpenCodex now completes that transition by keeping the managed root `openai_base_url` override beside the `[model_providers.opencodex]` table. Codex merges the override onto its built-in `openai` provider, so those conversations keep reaching the proxy without being relabeled and no rollout byte or thread row is touched. Only a routing form that requires the `x-opencodex-api-key` admission header still refuses, because Codex's built-in provider cannot carry that header; its message names the two settings that resolve it — route Codex through the loopback listener so the override can be retained, or set `syncResumeHistory` to `false` to accept that those conversations resume against Codex's own OpenAI endpoint. +### Remote thread-list provider filters + +Provider-table routing changes the default provider id for new conversations to `opencodex` while +history that cannot safely be relabeled may remain tagged `openai`. Some native app-server/mobile +versions treat an omitted `thread/list.modelProviders` filter as the current default provider only, +so those existing conversations can disappear from that remote list even though their database row +and rollout are intact. A compatible list client can send `modelProviders: []` to request all +providers. OpenCodex cannot rewrite that RPC because the remote client talks directly to Codex's +native app-server rather than the inference proxy. + +`ocx sync` and `ocx start` include the warning when they apply a provider-table route. If a +user-owned root URL sends the command down the no-routing branch, the CLI omits the warning. In +client-compaction mode, the CLI can retain that URL, apply the `opencodex` provider table, and +include the warning. The dashboard shows a separate preference hint when either setting is enabled. +It appears once if both settings are enabled, regardless of the root URL. The hint reports enabled +preferences; it does not mean Authless Desktop is effective on the current route. Authless Desktop +applies only to effective loopback authless routing and is ignored for remote-client routing or +listeners that require an admission header. The warning is not a migration: OpenCodex does not edit +provider tags merely to influence a client-side list filter. Verify the conversation in native Codex +and the app-server/client version; do not rewrite paginated history to make a remote list include it. + Do not rewrite an active paginated rollout or thread row to migrate those conversations yourself. Close the affected conversation before any recovery, and report the exact error and versions without uploading private history. A backup or a successful script alone does not prove the conversation is visible again. Check the restored conversation in Codex after reopening. ## Experimental native mid-turn steering diff --git a/gui/src/i18n/de.ts b/gui/src/i18n/de.ts index 2b83d98306a..c9dfd974a3c 100644 --- a/gui/src/i18n/de.ts +++ b/gui/src/i18n/de.ts @@ -3141,6 +3141,7 @@ export const de: Record = { "dash.codexDesktopAuthlessHint": "Standardmäßig aus. Überspringt die separate Desktop-Anmeldung bei geeigneten lokalen Verbindungen. Zugangsdaten für den Anbieter bleiben erforderlich. Codex nach einer Änderung neu starten. Kontogebundene Desktop-Funktionen können fehlen.", "dash.codexClientCompaction": "Clientseitige Komprimierung verwenden", "dash.codexClientCompactionHint": "Standardmäßig aus; nur für authentifiziertes Loopback-Routing. Künftige Komprimierungen speichern portable Klartext-Zusammenfassungen, während das OpenCodeX-Provider-Routing und die V2-Subagent-Zustellung aktiv bleiben; der konfigurierte Anbieter kann sie verarbeiten und Kontingent verbrauchen. Vorhandene ocx1-Verläufe müssen weiterhin wiederhergestellt werden. Codex nach einer Änderung neu starten.", + "dash.codexRemoteHistoryHint": "Provider-Tabellen können vorhandene Threads mit openai-Kennung in manchen mobilen Remote-Listen ausblenden. Der Verlauf wird nicht gelöscht. Der Remote-Client muss alle Provider auflisten; dieser Schalter korrigiert dessen Filter nicht.", "models.newPolicyGlobal": "Neue Modelle zunächst deaktivieren", "models.newPolicyProvider": "Richtlinie für neue Modelle", "models.fastProvider": "Fast-Modus", "models.fastProviderHint": "Verbraucht Nutzungsguthaben zum doppelten Preis", "models.fastEnabled": "Fast-Modus an", "models.fastDisabled": "Fast-Modus aus", "models.fastSaveFailed": "Fast-Modus konnte nicht gespeichert werden", "models.newPolicy_inherit": "Übernehmen", "models.newPolicy_off": "Aus", "models.newPolicy_on": "An", "models.newBadge": "NEU", "models.newCount": "{count} neu, aus", diff --git a/gui/src/i18n/en.ts b/gui/src/i18n/en.ts index dad65cfc6dd..d5eeb2e9273 100644 --- a/gui/src/i18n/en.ts +++ b/gui/src/i18n/en.ts @@ -718,6 +718,7 @@ export const en = { "dash.codexDesktopAuthlessHint": "Off by default. Skip the separate Desktop sign-in for eligible local connections. Upstream credentials are still required. Restart Codex after changing this setting. Account-gated Desktop features may be unavailable.", "dash.codexClientCompaction": "Use client-side compaction", "dash.codexClientCompactionHint": "Off by default; authenticated loopback only. Future compactions store portable plaintext summaries while OpenCodeX and V2 provider routing stay active; the configured provider may process them and consume quota. History is left untouched, and existing threads keep routing through the proxy via the openai_base_url override OpenCodeX manages; if you set that line yourself it is kept, and those threads follow your destination instead. Existing ocx1 history stays recoverable; recover a thread separately only before replaying it in native Codex. Restart Codex after changing this setting.", + "dash.codexRemoteHistoryHint": "Provider-table routing can hide existing openai-tagged threads in some mobile remote lists. History is not deleted. The remote client must list all providers; this switch does not repair that client filter.", "models.v2Conflict": "[agents] max_threads is set — codex will refuse to start; remove it from config.toml", "models.v2Applied": "Sub-agent mode updated — applies to new sessions (restart the Codex app to refresh the picker)", "models.v2ThreadsLabel": "Max threads", diff --git a/gui/src/i18n/fr.ts b/gui/src/i18n/fr.ts index ce302bb3e37..5ca92bf3726 100644 --- a/gui/src/i18n/fr.ts +++ b/gui/src/i18n/fr.ts @@ -703,6 +703,7 @@ export const fr: Record = { "dash.codexDesktopAuthlessHint": "Désactivé par défaut. Ignore la connexion Desktop séparée pour les connexions locales admissibles. Les identifiants du fournisseur restent nécessaires. Redémarrez Codex après toute modification. Certaines fonctions Desktop liées au compte peuvent être indisponibles.", "dash.codexClientCompaction": "Utiliser la compaction côté client", "dash.codexClientCompactionHint": "Désactivé par défaut, uniquement pour le routage loopback authentifié. Les compactages futurs stockent des résumés portables en texte clair tout en conservant le routage OpenCodeX/V2 ; le fournisseur configuré peut les traiter et consommer son quota. L'historique ocx1 existant doit toujours être restauré. Redémarrez Codex après modification.", + "dash.codexRemoteHistoryHint": "Le routage par table de fournisseurs peut masquer les fils existants marqués openai dans certaines listes mobiles distantes. L’historique n’est pas supprimé. Le client distant doit lister tous les fournisseurs ; ce réglage ne corrige pas son filtre.", "models.v2Conflict": "[agents] max_threads est défini — codex refusera de démarrer ; supprimez-le de config.toml", "models.v2Applied": "Mode sous-agent mis à jour — s’applique aux nouvelles sessions (redémarrez l’application Codex pour actualiser le sélecteur)", "models.v2ThreadsLabel": "Nombre maximal de fils", diff --git a/gui/src/i18n/ja.ts b/gui/src/i18n/ja.ts index 938ee65ec43..67b3a4973f3 100644 --- a/gui/src/i18n/ja.ts +++ b/gui/src/i18n/ja.ts @@ -3163,6 +3163,7 @@ export const ja: Record = { "dash.codexDesktopAuthlessHint": "既定ではオフです。対象のローカル接続で Desktop の個別ログインを省略します。上流プロバイダーの認証情報は引き続き必要です。変更後は Codex を再起動してください。アカウントに依存する Desktop 機能が利用できない場合があります。", "dash.codexClientCompaction": "クライアント側コンパクションを使用", "dash.codexClientCompactionHint": "既定ではオフで、認証済みループバックルーティング専用です。今後のコンパクションは、OpenCodeX と V2 プロバイダーのルーティングを維持したまま移植可能な平文要約を保存します。設定済みプロバイダーが要約を処理し、割り当てを消費する場合があります。既存の ocx1 履歴は別途復旧が必要です。変更後は Codex を再起動してください。", + "dash.codexRemoteHistoryHint": "プロバイダーテーブル方式では、一部のモバイルリモート一覧で既存の openai タグ付きスレッドが表示されない場合があります。履歴は削除されていません。リモートクライアントは全プロバイダーを一覧取得する必要があり、このスイッチはそのフィルターを修正しません。", "models.newPolicyGlobal": "新しいモデルを無効で追加", "models.newPolicyProvider": "新しいモデルのポリシー", "models.fastProvider": "Fast モード", "models.fastProviderHint": "使用クレジットを 2 倍の料金で消費します", "models.fastEnabled": "Fast モードをオンにしました", "models.fastDisabled": "Fast モードをオフにしました", "models.fastSaveFailed": "Fast モードを保存できませんでした", "models.newPolicy_inherit": "継承", "models.newPolicy_off": "オフ", "models.newPolicy_on": "オン", "models.newBadge": "新着", "models.newCount": "新着 {count} 件、オフ", diff --git a/gui/src/i18n/ko.ts b/gui/src/i18n/ko.ts index 80b4af705c1..ecb3c2b1003 100644 --- a/gui/src/i18n/ko.ts +++ b/gui/src/i18n/ko.ts @@ -3163,6 +3163,7 @@ export const ko: Record = { "dash.codexDesktopAuthlessHint": "기본값은 꺼짐입니다. 지원되는 로컬 연결에서 별도의 Desktop 로그인을 건너뜁니다. 업스트림 인증 정보는 여전히 필요합니다. 변경 후 Codex를 다시 시작하세요. 계정에 연결된 Desktop 기능을 사용하지 못할 수 있습니다.", "dash.codexClientCompaction": "클라이언트 측 컴팩션 사용", "dash.codexClientCompactionHint": "기본값은 꺼짐이며 인증된 루프백 라우팅에만 적용됩니다. 향후 컴팩션은 OpenCodeX 및 V2 제공자 라우팅을 유지하면서 이식 가능한 평문 요약을 저장합니다. 설정된 제공자가 요약을 처리하고 할당량을 사용할 수 있습니다. 기록은 건드리지 않으며, 기존 스레드는 OpenCodeX가 관리하는 openai_base_url override를 통해 프록시 경로를 유지합니다. 그 줄을 직접 설정해 두셨다면 그대로 보존하므로 해당 스레드는 설정하신 목적지를 따릅니다. 기존 ocx1 기록은 그대로 복구할 수 있고, 네이티브 Codex에서 해당 스레드를 재개하기 전에만 별도로 복구하세요. 변경 후 Codex를 다시 시작하세요.", + "dash.codexRemoteHistoryHint": "제공자 테이블 방식에서는 일부 모바일 원격 목록에 기존 openai 태그 스레드가 보이지 않을 수 있습니다. 기록이 삭제된 것은 아닙니다. 원격 클라이언트가 모든 제공자를 조회해야 하며, 이 스위치는 해당 목록 필터를 수정하지 않습니다.", "models.newPolicyGlobal": "새 모델을 비활성화 상태로 추가", "models.newPolicyProvider": "새 모델 정책", "models.fastProvider": "Fast 모드", "models.fastProviderHint": "사용 크레딧을 2배 가격으로 소모합니다", "models.fastEnabled": "Fast 모드를 켰습니다", "models.fastDisabled": "Fast 모드를 껐습니다", "models.fastSaveFailed": "Fast 모드를 저장하지 못했습니다", "models.newPolicy_inherit": "상속", "models.newPolicy_off": "끔", "models.newPolicy_on": "켬", "models.newBadge": "신규", "models.newCount": "신규 {count}개, 꺼짐", diff --git a/gui/src/i18n/ru.ts b/gui/src/i18n/ru.ts index 2fcf9a11a33..16a41bf6975 100644 --- a/gui/src/i18n/ru.ts +++ b/gui/src/i18n/ru.ts @@ -3164,6 +3164,7 @@ export const ru: Record = { "dash.codexDesktopAuthlessHint": "По умолчанию выключено. Пропускает отдельный вход в Desktop для допустимых локальных подключений. Учётные данные провайдера по-прежнему нужны. После изменения перезапустите Codex. Функции Desktop, связанные с аккаунтом, могут быть недоступны.", "dash.codexClientCompaction": "Использовать сжатие на стороне клиента", "dash.codexClientCompactionHint": "По умолчанию выключено; только для аутентифицированной loopback-маршрутизации. Будущие сжатия сохраняют переносимые текстовые сводки, а маршрутизация OpenCodeX и V2 остаётся активной; настроенный провайдер может обрабатывать сводки и расходовать квоту. Существующую историю ocx1 всё равно нужно восстановить. После изменения перезапустите Codex.", + "dash.codexRemoteHistoryHint": "Таблица провайдеров может скрыть существующие диалоги с меткой openai в некоторых мобильных удалённых списках. История не удаляется. Удалённый клиент должен запрашивать все провайдеры; этот переключатель не исправляет его фильтр.", "models.newPolicyGlobal": "Добавлять новые модели выключенными", "models.newPolicyProvider": "Политика новых моделей", "models.fastProvider": "Режим Fast", "models.fastProviderHint": "Расходует кредиты использования по двойной цене", "models.fastEnabled": "Режим Fast включён", "models.fastDisabled": "Режим Fast выключен", "models.fastSaveFailed": "Не удалось сохранить режим Fast", "models.newPolicy_inherit": "Наследовать", "models.newPolicy_off": "Выкл.", "models.newPolicy_on": "Вкл.", "models.newBadge": "НОВАЯ", "models.newCount": "Новых: {count}, выкл.", diff --git a/gui/src/i18n/tr.ts b/gui/src/i18n/tr.ts index 4d7f56ccedd..b592e84752c 100644 --- a/gui/src/i18n/tr.ts +++ b/gui/src/i18n/tr.ts @@ -3164,6 +3164,7 @@ export const tr: Record = { "dash.codexDesktopAuthlessHint": "Varsayılan olarak kapalıdır. Uygun yerel bağlantılarda ayrı Desktop oturum açma adımını atlar. Sağlayıcı kimlik bilgileri yine gereklidir. Değişiklikten sonra Codex’i yeniden başlatın. Hesaba bağlı Desktop özellikleri kullanılamayabilir.", "dash.codexClientCompaction": "İstemci tarafı sıkıştırmayı kullan", "dash.codexClientCompactionHint": "Varsayılan olarak kapalıdır ve yalnızca kimliği doğrulanmış geri döngü yönlendirmesinde geçerlidir. Gelecekteki sıkıştırmalar, OpenCodeX ve V2 sağlayıcı yönlendirmesi etkin kalırken taşınabilir düz metin özetleri kaydeder; yapılandırılmış sağlayıcı bunları işleyip kotasını tüketebilir. Mevcut ocx1 geçmişi yine ayrıca kurtarılmalıdır. Değişiklikten sonra Codex’i yeniden başlatın.", + "dash.codexRemoteHistoryHint": "Sağlayıcı tablosuyla yönlendirme, mevcut openai etiketli konuşmaları bazı mobil uzak listelerde gizleyebilir. Geçmiş silinmez. Uzak istemci tüm sağlayıcıları listelemelidir; bu anahtar istemcinin filtresini düzeltmez.", "models.newPolicyGlobal": "Yeni modeller devre dışı başlasın", "models.newPolicyProvider": "Yeni model ilkesi", "models.fastProvider": "Fast modu", "models.fastProviderHint": "Kullanım kredilerini 2 kat fiyatla harcar", "models.fastEnabled": "Fast modu açık", "models.fastDisabled": "Fast modu kapalı", "models.fastSaveFailed": "Fast modu kaydedilemedi", "models.newPolicy_inherit": "Devral", "models.newPolicy_off": "Kapalı", "models.newPolicy_on": "Açık", "models.newBadge": "YENİ", "models.newCount": "{count} yeni, kapalı", diff --git a/gui/src/i18n/vi.ts b/gui/src/i18n/vi.ts index 5dba045198c..4be42b9ebc6 100644 --- a/gui/src/i18n/vi.ts +++ b/gui/src/i18n/vi.ts @@ -703,6 +703,7 @@ export const vi: Record = { "dash.codexDesktopAuthlessHint": "Mặc định tắt. Bỏ qua bước đăng nhập Desktop riêng cho các kết nối cục bộ hợp lệ. Vẫn cần thông tin xác thực upstream. Khởi động lại Codex sau khi thay đổi cài đặt này. Các tính năng Desktop yêu cầu tài khoản có thể không khả dụng.", "dash.codexClientCompaction": "Sử dụng tính năng thu gọn phía client (client-side compaction)", "dash.codexClientCompactionHint": "Mặc định tắt; chỉ dành cho authenticated loopback. Các compactions trong tương lai sẽ lưu trữ các bản tóm tắt văn bản thuần trong khi OpenCodeX và V2 provider routing vẫn hoạt động; provider được cấu hình có thể xử lý chúng và tiêu tốn quota. Lịch sử được giữ nguyên, và các luồng (threads) hiện tại tiếp tục định tuyến qua proxy thông qua ghi đè openai_base_url mà OpenCodeX quản lý; nếu bạn tự thiết lập dòng đó, nó sẽ được giữ lại và các luồng đó sẽ tuân theo đích đến của bạn thay thế. Lịch sử ocx1 hiện tại vẫn có thể khôi phục được; chỉ khôi phục một luồng riêng biệt trước khi phát lại (replay) trong Codex native. Khởi động lại Codex sau khi thay đổi cài đặt này.", + "dash.codexRemoteHistoryHint": "Định tuyến bằng bảng nhà cung cấp có thể ẩn các luồng mang nhãn openai hiện có khỏi một số danh sách từ xa trên điện thoại. Lịch sử không bị xóa. Client từ xa phải liệt kê mọi nhà cung cấp; công tắc này không sửa bộ lọc của client.", "models.v2Conflict": "[agents] max_threads đã được thiết lập — codex sẽ từ chối khởi động; hãy xoá nó khỏi config.toml", "models.v2Applied": "Chế độ agent con đã được cập nhật — áp dụng cho các phiên mới (khởi động lại ứng dụng Codex để làm mới picker)", "models.v2ThreadsLabel": "Luồng tối đa (Max threads)", diff --git a/gui/src/i18n/zh-TW.ts b/gui/src/i18n/zh-TW.ts index f8aee1a7a86..9285fe2a451 100644 --- a/gui/src/i18n/zh-TW.ts +++ b/gui/src/i18n/zh-TW.ts @@ -3127,6 +3127,7 @@ export const zhTW: Record = { "dash.codexDesktopAuthlessHint": "預設關閉。為符合條件的本機連線略過獨立的 Desktop 登入。仍需上游供應商憑證。變更後請重新啟動 Codex。依賴帳戶的 Desktop 功能可能無法使用。", "dash.codexClientCompaction": "使用用戶端壓縮", "dash.codexClientCompactionHint": "預設關閉,僅適用於已驗證的 loopback 路由。未來壓縮會儲存可攜的純文字摘要,同時保留 OpenCodeX 與 V2 提供方路由;已設定的提供方可能處理摘要並消耗其額度。既有 ocx1 歷程仍須另行復原。變更後請重新啟動 Codex。", + "dash.codexRemoteHistoryHint": "供應商表路由可能使既有的 openai 標記對話在部分行動版遠端清單中隱藏。歷程並未刪除。遠端用戶端必須列出所有供應商;此開關不會修正用戶端的清單篩選器。", "models.newPolicyGlobal": "新模型預設停用", "models.newPolicyProvider": "新模型策略", "models.fastProvider": "Fast 模式", "models.fastProviderHint": "以 2 倍價格消耗用量額度", "models.fastEnabled": "已開啟 Fast 模式", "models.fastDisabled": "已關閉 Fast 模式", "models.fastSaveFailed": "無法儲存 Fast 模式", "models.newPolicy_inherit": "繼承", "models.newPolicy_off": "關閉", "models.newPolicy_on": "開啟", "models.newBadge": "新增", "models.newCount": "{count} 個新增,已關閉", diff --git a/gui/src/i18n/zh.ts b/gui/src/i18n/zh.ts index 0e740414da5..d30a30fe4e1 100644 --- a/gui/src/i18n/zh.ts +++ b/gui/src/i18n/zh.ts @@ -3162,6 +3162,7 @@ export const zh: Record = { "dash.codexDesktopAuthlessHint": "默认关闭。为符合条件的本地连接跳过单独的 Desktop 登录。仍需上游提供商凭据。更改后请重启 Codex。依赖账户的 Desktop 功能可能不可用。", "dash.codexClientCompaction": "使用客户端压缩", "dash.codexClientCompactionHint": "默认关闭,仅适用于已认证的 loopback 路由。未来压缩会保存可移植的明文摘要,同时保留 OpenCodeX 与 V2 提供方路由;已配置的提供方可能处理摘要并消耗其额度。已有 ocx1 历史仍需单独恢复。更改后请重启 Codex。", + "dash.codexRemoteHistoryHint": "提供商表路由可能使已有的 openai 标签会话在部分移动端远程列表中隐藏。历史记录并未删除。远程客户端必须列出所有提供商;此开关不会修复客户端的列表过滤器。", "models.newPolicyGlobal": "新模型默认停用", "models.newPolicyProvider": "新模型策略", "models.fastProvider": "Fast 模式", "models.fastProviderHint": "按 2 倍价格消耗用量额度", "models.fastEnabled": "已开启 Fast 模式", "models.fastDisabled": "已关闭 Fast 模式", "models.fastSaveFailed": "无法保存 Fast 模式", "models.newPolicy_inherit": "继承", "models.newPolicy_off": "关闭", "models.newPolicy_on": "开启", "models.newBadge": "新增", "models.newCount": "{count} 个新增,已关闭", diff --git a/gui/src/pages/dashboard-overview-sections.tsx b/gui/src/pages/dashboard-overview-sections.tsx index 8ea50ba758c..f3e61161b88 100644 --- a/gui/src/pages/dashboard-overview-sections.tsx +++ b/gui/src/pages/dashboard-overview-sections.tsx @@ -517,6 +517,7 @@ export function DashboardSidecarPanels({ d }: { d: Dash }) {
{t("dash.codexDesktopAuthless")}
{t("dash.codexDesktopAuthlessHint")}
+ {settings?.codexDesktopAuthless &&
{t("dash.codexRemoteHistoryHint")}
} {settings?.catalogRefreshPending &&
{t("codexAuth.catalogRefreshPending")}
}