From 87f970fd997b6d9749c3e3718a488f61cfd4ad81 Mon Sep 17 00:00:00 2001 From: Robin Bially <7304732+RobinBially@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:35:55 +0200 Subject: [PATCH 01/13] feat(memory): route Codex memory phases to a chosen model Codex writes memories in two phases: an extract pass per finished session and a consolidation pass that merges those notes into the memory files. Each phase asks for its own model, so on a routed setup both land on the OpenAI account while the rest of the traffic runs on a configured provider. Add a `memoryModels` config block and a Memory routing panel that pick a model and an optional reasoning effort per phase. The phases are recognized from Codex's own turn metadata (`request_kind: "memory"` for extract, `thread_source: "memory_consolidation"` plus the sub-agent header for consolidation) and never from the model id, because the extract phase shares its model with Codex's helper calls. A configured phase wins over the shadow-call intercept; helper calls stay untouched. --- .../docs/reference/configuration/server.md | 47 +++ gui/src/components/MemoryModelsPanel.tsx | 185 +++++++++++ gui/src/i18n/de.ts | 17 + gui/src/i18n/en.ts | 17 + gui/src/i18n/fr.ts | 17 + gui/src/i18n/ja.ts | 17 + gui/src/i18n/ko.ts | 17 + gui/src/i18n/ru.ts | 17 + gui/src/i18n/tr.ts | 17 + gui/src/i18n/vi.ts | 17 + gui/src/i18n/zh-TW.ts | 17 + gui/src/i18n/zh.ts | 17 + gui/src/pages/dashboard-overview-panels.tsx | 2 + scripts/test-layout/layout.json | 1 + src/config/diagnostics.ts | 5 + src/config/load-degrade.ts | 12 + src/config/schema/config-schema.ts | 4 + src/config/schema/leaf-validators.ts | 15 + src/server/management/config-routes.ts | 22 +- src/server/responses/core-normalize.ts | 16 + src/server/responses/core-options.ts | 2 + src/server/responses/memory-models.ts | 158 +++++++++ src/server/responses/request-prepare.ts | 69 +++- .../responses/shadow-target-availability.ts | 21 +- src/types/config.ts | 21 ++ src/types/request.ts | 6 + tests/fixtures/test-layout-expected.json | 1 + .../responses/responses-memory-models.test.ts | 304 ++++++++++++++++++ 28 files changed, 1049 insertions(+), 12 deletions(-) create mode 100644 gui/src/components/MemoryModelsPanel.tsx create mode 100644 src/server/responses/memory-models.ts create mode 100644 tests/responses/responses-memory-models.test.ts diff --git a/docs-site/src/content/docs/reference/configuration/server.md b/docs-site/src/content/docs/reference/configuration/server.md index c8e9f967ffb..4ffd03d41e3 100644 --- a/docs-site/src/content/docs/reference/configuration/server.md +++ b/docs-site/src/content/docs/reference/configuration/server.md @@ -41,6 +41,7 @@ runs helper features around provider requests. | `resetCreditAutoRedeem?` | `{ enabled?: boolean; leadTimeMinutes?: number }` | off | Opt-in: redeem the main Codex account's soonest-expiring reset credit `leadTimeMinutes` (1–60, default 10) before it expires. Every attempt re-reads the upstream credit list first and skips when the credit is gone (for example, redeemed by hand); the `redeem_request_id` is journaled in `$OPENCODEX_HOME/reset-credit-auto-redeem.json` before the call so a crash replays the same idempotent request instead of spending a second credit. Servers sharing this configuration directory coordinate reservations and settlements so one process does not replace another's request record. Logs carry a hashed account key only. | | `syncResumeHistory?` | `boolean` | `true` | Reversible Codex App history compatibility. Original metadata is backed up and restored by `ocx stop` / `ocx restore`. | | `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | Redirect recognized Codex helper/shadow calls to a chosen model while preserving the request's configured reasoning effort. The default source prefixes are `gpt-6-luna` and `gpt-5.6-luna`; older clients through 0.144.x used `gpt-5.4-mini`, which `sourceModels` can restore. | +| `memoryModels?` | `{ extract?: { model: string; reasoningEffort?: string }; consolidation?: { model: string; reasoningEffort?: string } }` | off | Route Codex's two memory phases to a chosen model, with an optional reasoning effort per phase. See [Memory routing](#memory-routing). | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | on when usable | Web-search sidecar options. | | `visionSidecar?` | `OcxVisionSidecarConfig` | on when usable | Image-description sidecar options. | | `images?` | `OcxImagesConfig` | automatic OpenAI selection | Standalone Images relay options for Codex `image_gen`. | @@ -658,6 +659,52 @@ caller's credential does not cross to the other provider. The selected model mus input size and content. Restart the proxy after editing `config.json` by hand. Dashboard saves apply immediately. +## Memory routing + +In **Dashboard → Overview → Memory routing**, choose a model and an optional reasoning effort for +each of Codex's two memory phases, then click **Save**. Select **Off — Codex default** and save to +remove the override. Changes apply to the next memory request without restarting the proxy. + +Set `memoryModels` in OpenCodex `config.json` to route those requests. Both phases keep Codex's own +model while the block is omitted, and the phases are independent: configuring one leaves the other +alone. + +```json +{ + "memoryModels": { + "extract": { "model": "provider/model-id", "reasoningEffort": "low" }, + "consolidation": { "model": "provider/model-id", "reasoningEffort": "medium" } + } +} +``` + +`extract` is the pass that summarizes one finished session into a raw memory; `consolidation` is the +single agent run that merges those raw memories into the files under `$CODEX_HOME/memories`. +`model` accepts native model IDs, provider-qualified model IDs, and configured combos. +`reasoningEffort` is optional; omit it to keep the effort Codex asked for. Supported declarations are +`none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`, and `ultra`. Codex hard-codes `low` for +extract and `medium` for consolidation, so a configured effort replaces that value. + +OpenCodex recognizes these requests from Codex's own turn metadata only: `request_kind: "memory"` in +the `x-codex-turn-metadata` header marks an extract pass, and `thread_source: +"memory_consolidation"` (plus the `x-openai-subagent: memory_consolidation` header on the +consolidation pass) marks the consolidation thread. The model id is deliberately not a signal: the +extract pass runs on the same helper model Codex uses for titles and commit messages, so a +model-based rule would also capture ordinary helper calls. Missing, malformed, or conflicting +metadata does not activate the override; when several copies of the metadata are supplied they must +name the same phase. WebSocket requests use each frame's metadata rather than the connection's +earlier handshake metadata. + +A configured phase wins when `shadowCallIntercept` would match the same request. A phase left off +keeps its current routing, which includes the shadow-call intercept: extract runs on a helper model +id, so an enabled intercept already covers it. The selected model's provider receives the session +text Codex summarizes for memory, including sessions that normally run on another provider; the +dashboard panel states this next to the model pickers. Without a choice, memory requests reach your +OpenAI account like any other native model. A phase whose target stopped resolving — the provider is +disabled or deleted, or its combo no longer exists — fails that memory call with `409` and error code +`memory_model_target_unavailable` instead of falling back to the default provider. The request log +names the phase (`memory-extract` or `memory-consolidation`) as the routing reason. Restart the +proxy after editing `config.json` by hand. Dashboard saves apply immediately. ## Shadow calls Codex uses small helper models for tasks such as titles and commit messages. Enable diff --git a/gui/src/components/MemoryModelsPanel.tsx b/gui/src/components/MemoryModelsPanel.tsx new file mode 100644 index 00000000000..dcd00d5e478 --- /dev/null +++ b/gui/src/components/MemoryModelsPanel.tsx @@ -0,0 +1,185 @@ +import { useCallback, useEffect, useRef, useState } from "react"; +import { useT, type TKey } from "../i18n/shared"; +import { IconAlert } from "../icons"; +import { Select, Tooltip } from "../ui"; +import { createBoundedFetch } from "../bounded-fetch"; +import { requireJson, type ModelInfo } from "../pages/dashboard-shared"; +import { formatNamespacedModelId } from "../provider-icons"; + +type Phase = "extract" | "consolidation"; +interface PhaseSetting { model?: string; reasoningEffort?: string } +type Settings = { extract?: PhaseSetting; consolidation?: PhaseSetting }; + +const EFFORTS = ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]; + +/** + * Read the persisted phases. A phase without a model is "Off", so it is dropped rather than kept + * as an empty row: that is also the shape the PUT sends back for it. + */ +function readSettings(payload: { memoryModels?: unknown }): Settings { + const value = payload.memoryModels; + if (value == null) return {}; + if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("invalid settings"); + const out: Settings = {}; + for (const phase of ["extract", "consolidation"] as const) { + const raw = (value as Record)[phase]; + if (raw === undefined) continue; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("invalid phase"); + const model = "model" in raw && typeof raw.model === "string" ? raw.model.trim() : ""; + if (!model) throw new Error("invalid model"); + const effort = "reasoningEffort" in raw ? raw.reasoningEffort : undefined; + if (effort !== undefined && (typeof effort !== "string" || !EFFORTS.includes(effort))) throw new Error("invalid effort"); + out[phase] = { model, ...(effort ? { reasoningEffort: effort } : {}) }; + } + return out; +} + +export default function MemoryModelsPanel(props: { apiBase: string; models: ModelInfo[] }) { + return ; +} + +function MemoryModelsControls({ apiBase, models }: { apiBase: string; models: ModelInfo[] }) { + const t = useT(); + const [saved, setSaved] = useState(undefined); + const [extractModel, setExtractModel] = useState(""); + const [extractEffort, setExtractEffort] = useState(""); + const [consolidationModel, setConsolidationModel] = useState(""); + const [consolidationEffort, setConsolidationEffort] = useState(""); + const [busy, setBusy] = useState(false); + const [loadError, setLoadError] = useState(false); + const [feedback, setFeedback] = useState<"saved" | "failed" | null>(null); + const active = useRef(false); + const pending = useRef | null>(null); + + const accept = useCallback((value: Settings) => { + setSaved(value); + setExtractModel(value.extract?.model ?? ""); + setExtractEffort(value.extract?.reasoningEffort ?? ""); + setConsolidationModel(value.consolidation?.model ?? ""); + setConsolidationEffort(value.consolidation?.reasoningEffort ?? ""); + }, []); + + const load = useCallback(async () => { + if (pending.current) return; + const request = createBoundedFetch(15_000); + pending.current = request; + setLoadError(false); + try { + const response = await fetch(`${apiBase}/api/settings`, { signal: request.signal }); + const value = readSettings(await requireJson(response)); + if (active.current && pending.current === request) accept(value); + } catch { + if (active.current && pending.current === request) setLoadError(true); + } finally { + request.clear(); + if (pending.current === request) pending.current = null; + } + }, [apiBase, accept]); + + useEffect(() => { + active.current = true; + const timer = window.setTimeout(() => { void load(); }, 0); + return () => { + window.clearTimeout(timer); + active.current = false; + pending.current?.controller.abort(); + pending.current?.clear(); + pending.current = null; + }; + }, [load]); + + const phasePayload = (model: string, effort: string) => (model + ? { model, ...(effort ? { reasoningEffort: effort } : {}) } + : undefined); + + const save = async () => { + if (pending.current || saved === undefined) return; + const request = createBoundedFetch(15_000); + pending.current = request; + setBusy(true); + setFeedback(null); + const extract = phasePayload(extractModel, extractEffort); + const consolidation = phasePayload(consolidationModel, consolidationEffort); + try { + const response = await fetch(`${apiBase}/api/settings`, { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + // Null clears the whole block; a phase left at "Off" is simply absent. + memoryModels: extract || consolidation + ? { ...(extract ? { extract } : {}), ...(consolidation ? { consolidation } : {}) } + : null, + }), + signal: request.signal, + }); + const value = readSettings(await requireJson(response)); + if (active.current && pending.current === request) { + accept(value); + setFeedback("saved"); + } + } catch { + if (active.current && pending.current === request) setFeedback("failed"); + } finally { + request.clear(); + if (active.current && pending.current === request) setBusy(false); + if (pending.current === request) pending.current = null; + } + }; + + const options = [{ value: "", label: t("memoryModels.off") }, + ...[...new Set([...models.map(item => item.namespaced), + ...[extractModel, consolidationModel].filter(Boolean)])] + .map(value => ({ value, label: formatNamespacedModelId(value, t) }))]; + const effortOptions = [{ value: "", label: t("memoryModels.defaultEffort") }, + ...EFFORTS.map(value => ({ value, label: t(`models.reasoningEffort.${value}` as TKey) }))]; + const disabled = busy || saved === undefined || loadError; + const dirty = extractModel !== (saved?.extract?.model ?? "") + || extractEffort !== (saved?.extract?.reasoningEffort ?? "") + || consolidationModel !== (saved?.consolidation?.model ?? "") + || consolidationEffort !== (saved?.consolidation?.reasoningEffort ?? ""); + const info = t("memoryModels.info"); + + const row = (phase: Phase, model: string, effort: string, setModel: (value: string) => void, setEffort: (value: string) => void) => ( +
+
+
{t(`memoryModels.${phase}` as TKey)}
+
{t(`memoryModels.${phase}Hint` as TKey)}
+
+
+ { setEffort(value); setFeedback(null); }} /> +
+
+ ); + + return ( +
+
+ {t("memoryModels.title")}{" "} + + ⓘ + +
+
{t("memoryModels.description")}
+ {row("extract", extractModel, extractEffort, setExtractModel, setExtractEffort)} + {row("consolidation", consolidationModel, consolidationEffort, setConsolidationModel, setConsolidationEffort)} +
+
{t("memoryModels.dataNotice")}
+ +
+ {(extractModel || consolidationModel) &&
+ {t("memoryModels.accountNotice")} +
} + {loadError &&
{t("memoryModels.loadFailed")}
} + {feedback === "failed" &&
{t("memoryModels.saveFailed")}
} + {feedback === "saved" &&
{t("memoryModels.saved")}
} +
+ ); +} + diff --git a/gui/src/i18n/de.ts b/gui/src/i18n/de.ts index c9dfd974a3c..a9f647f443d 100644 --- a/gui/src/i18n/de.ts +++ b/gui/src/i18n/de.ts @@ -422,6 +422,23 @@ export const de: Record = { "compactionRouting.loadFailed": "Komprimierungseinstellungen konnten nicht geladen werden.", "compactionRouting.saved": "Komprimierungseinstellungen gespeichert.", "compactionRouting.saveFailed": "Speichern fehlgeschlagen. Deine Änderungen sind noch vorhanden; versuche es erneut.", + "memoryModels.title": "Memory-Routing", + "memoryModels.description": "Codex schreibt Memories im Hintergrund, nachdem eine Sitzung endet. Wähle das Modell für jeden Schritt — oder lass Codex selbst wählen.", + "memoryModels.infoLabel": "Was sind Extract und Consolidation?", + "memoryModels.info": "Codex macht Memory in zwei Schritten. Extract liest eine beendete Sitzung und notiert, was passiert ist: ein Notizzettel pro Sitzung, also viele kleine Aufrufe. Consolidation nimmt diese Zettel und schreibt sie in die Memory-Dateien, die Codex am Anfang deiner nächsten Sitzungen liest. Läuft selten, bearbeitet aber Dateien. Jeder Schritt fragt sein Modell selbst an, deshalb stehen sie hier getrennt.", + "memoryModels.extract": "Extraktion", + "memoryModels.extractHint": "Fasst jede beendete Sitzung zu einem Raw Memory zusammen. Läuft einmal pro Sitzung.", + "memoryModels.consolidation": "Konsolidierung", + "memoryModels.consolidationHint": "Führt die Raw Memories in die Memory-Dateien zusammen, die Codex später liest. Läuft selten und bearbeitet Dateien.", + "memoryModels.model": "Modell", + "memoryModels.effort": "Reasoning-Aufwand", + "memoryModels.off": "Aus — Codex-Standard", + "memoryModels.defaultEffort": "Codex-Standard", + "memoryModels.dataNotice": "Das gewählte Modell erhält den Sitzungstext, den Codex für das Memory zusammenfasst.", + "memoryModels.accountNotice": "Ohne Auswahl hier gehen Memory-Anfragen wie jedes andere native Modell an dein OpenAI-Konto.", + "memoryModels.loadFailed": "Memory-Einstellungen konnten nicht geladen werden.", + "memoryModels.saved": "Memory-Einstellungen gespeichert.", + "memoryModels.saveFailed": "Speichern fehlgeschlagen. Deine Änderungen stehen noch da; versuch es erneut.", "dash.shadowCallIntercept": "Shadow-Call-Abfangen", "dash.shadowCallInterceptHint": "Fängt die Hintergrund-Hilfsaufrufe der Codex-App ({models}) ab und leitet sie an das gewählte Modell um.", "dash.shadowCallWarning": "⚠ Bei Aktivierung werden ALLE Anfragen an {models} durch das gewählte Modell ersetzt.", diff --git a/gui/src/i18n/en.ts b/gui/src/i18n/en.ts index d5eeb2e9273..bc64b885c86 100644 --- a/gui/src/i18n/en.ts +++ b/gui/src/i18n/en.ts @@ -440,6 +440,23 @@ export const en = { "compactionRouting.loadFailed": "Could not load compaction settings.", "compactionRouting.saved": "Compaction settings saved.", "compactionRouting.saveFailed": "Could not save. Your changes are still here; try again.", + "memoryModels.title": "Memory routing", + "memoryModels.description": "Codex writes memories in the background after a session ends. Pick the model each step uses, or leave it on Codex's own choice.", + "memoryModels.infoLabel": "What are Extract and Consolidation?", + "memoryModels.info": "Codex turns finished sessions into memory in two steps. Extract reads one finished session and jots down what happened: one note per session, so it makes many small calls. Consolidation takes those notes and writes them into the memory files Codex reads at the start of your next sessions. It runs rarely, but it edits files. Each step asks for its own model, which is why they are listed separately here.", + "memoryModels.extract": "Extract", + "memoryModels.extractHint": "Summarizes each finished session into a raw memory. Runs once per session.", + "memoryModels.consolidation": "Consolidation", + "memoryModels.consolidationHint": "Merges the raw memories into the memory files Codex reads later. Runs rarely and edits files.", + "memoryModels.model": "Model", + "memoryModels.effort": "Reasoning effort", + "memoryModels.off": "Off — Codex default", + "memoryModels.defaultEffort": "Codex default", + "memoryModels.dataNotice": "The chosen model receives the session text Codex summarizes for memory.", + "memoryModels.accountNotice": "Without a choice here, memory requests go to your OpenAI account like any other native model.", + "memoryModels.loadFailed": "Could not load memory settings.", + "memoryModels.saved": "Memory settings saved.", + "memoryModels.saveFailed": "Could not save. Your changes are still here; try again.", "dash.shadowCallIntercept": "Shadow Call Intercept", "dash.shadowCallInterceptHint": "Intercepts Codex App's background helper calls ({models}) for title generation and commit messages and redirects them to your chosen model.", "dash.shadowCallWarning": "⚠ When enabled, ALL requests for {models} will be replaced with the selected model.", diff --git a/gui/src/i18n/fr.ts b/gui/src/i18n/fr.ts index 5ca92bf3726..eecf4748775 100644 --- a/gui/src/i18n/fr.ts +++ b/gui/src/i18n/fr.ts @@ -430,6 +430,23 @@ export const fr: Record = { "compactionRouting.loadFailed": "Impossible de charger les paramètres de compaction.", "compactionRouting.saved": "Paramètres de compaction enregistrés.", "compactionRouting.saveFailed": "Échec de l’enregistrement. Vos modifications sont conservées ; réessayez.", + "memoryModels.title": "Routage de la mémoire", + "memoryModels.description": "Codex écrit les mémoires en arrière-plan, une fois la session terminée. Choisissez le modèle de chaque étape, ou laissez Codex décider.", + "memoryModels.infoLabel": "Que sont Extract et Consolidation ?", + "memoryModels.info": "Codex transforme les sessions terminées en mémoire en deux étapes. Extract lit une session terminée et note ce qui s'y est passé : une fiche par session, donc beaucoup de petits appels. Consolidation reprend ces fiches et les écrit dans les fichiers de mémoire que Codex lit au début de vos sessions suivantes. Elle passe rarement, mais elle modifie des fichiers. Chaque étape demande son propre modèle, d'où ces deux lignes.", + "memoryModels.extract": "Extraction", + "memoryModels.extractHint": "Résume chaque session terminée en une mémoire brute. Une fois par session.", + "memoryModels.consolidation": "Consolidation", + "memoryModels.consolidationHint": "Fusionne les mémoires brutes dans les fichiers que Codex lit ensuite. Passe rarement et modifie des fichiers.", + "memoryModels.model": "Modèle", + "memoryModels.effort": "Effort de raisonnement", + "memoryModels.off": "Désactivé — valeur Codex", + "memoryModels.defaultEffort": "Valeur Codex", + "memoryModels.dataNotice": "Le modèle choisi reçoit le texte de session que Codex résume pour la mémoire.", + "memoryModels.accountNotice": "Sans choix ici, les requêtes de mémoire vont vers votre compte OpenAI comme tout autre modèle natif.", + "memoryModels.loadFailed": "Impossible de charger les réglages de mémoire.", + "memoryModels.saved": "Réglages de mémoire enregistrés.", + "memoryModels.saveFailed": "Échec de l'enregistrement. Vos modifications sont conservées ; réessayez.", "dash.shadowCallIntercept": "Interception des appels fantômes", "dash.shadowCallInterceptHint": "Intercepte les appels auxiliaires en arrière-plan de l’application Codex ({models}) pour générer les titres et les messages de commit, puis les redirige vers le modèle choisi.", "dash.shadowCallWarning": "⚠ Lorsque cette option est activée, TOUTES les requêtes destinées à {models} sont remplacées par le modèle sélectionné.", diff --git a/gui/src/i18n/ja.ts b/gui/src/i18n/ja.ts index 67b3a4973f3..16561f977d1 100644 --- a/gui/src/i18n/ja.ts +++ b/gui/src/i18n/ja.ts @@ -431,6 +431,23 @@ export const ja: Record = { "compactionRouting.loadFailed": "圧縮設定を読み込めませんでした。", "compactionRouting.saved": "圧縮設定を保存しました。", "compactionRouting.saveFailed": "保存できませんでした。変更内容は保持されています。再試行してください。", + "memoryModels.title": "メモリルーティング", + "memoryModels.description": "Codex はセッション終了後にバックグラウンドでメモリを書き込みます。各段階で使うモデルを選ぶか、Codex の既定のままにします。", + "memoryModels.infoLabel": "Extract と Consolidation とは?", + "memoryModels.info": "Codex は終了したセッションを 2 段階でメモリにします。Extract は終了したセッションを 1 つ読み、起きたことを書き留めます。セッションごとにメモ 1 枚、つまり小さな呼び出しがたくさん発生します。Consolidation はそのメモをまとめ、次のセッションの開始時に Codex が読むメモリファイルへ書き込みます。めったに動きませんが、ファイルを編集します。各段階が自分のモデルを要求するため、ここでは別々に表示しています。", + "memoryModels.extract": "抽出", + "memoryModels.extractHint": "終了したセッションごとに生のメモリへ要約します。セッションごとに 1 回動きます。", + "memoryModels.consolidation": "統合", + "memoryModels.consolidationHint": "生のメモリを、Codex が後で読むメモリファイルへ統合します。めったに動かず、ファイルを編集します。", + "memoryModels.model": "モデル", + "memoryModels.effort": "推論の強さ", + "memoryModels.off": "オフ — Codex の既定", + "memoryModels.defaultEffort": "Codex の既定", + "memoryModels.dataNotice": "選んだモデルには、Codex がメモリ用に要約するセッション本文が送られます。", + "memoryModels.accountNotice": "ここで選ばない場合、メモリのリクエストは他のネイティブモデルと同じく OpenAI アカウントへ送られます。", + "memoryModels.loadFailed": "メモリ設定を読み込めませんでした。", + "memoryModels.saved": "メモリ設定を保存しました。", + "memoryModels.saveFailed": "保存できませんでした。変更は残っています。もう一度お試しください。", "dash.shadowCallIntercept": "シャドウコール傍受", "dash.shadowCallInterceptHint": "Codex App のバックグラウンドヘルパー呼び出し({models}: タイトル生成、コミットメッセージ)を傍受し、選択したモデルにリダイレクトします。", "dash.shadowCallWarning": "⚠ オンにすると、{models} へのリクエストがすべて選択したモデルに置き換えられます。", diff --git a/gui/src/i18n/ko.ts b/gui/src/i18n/ko.ts index ecb3c2b1003..92b16d76277 100644 --- a/gui/src/i18n/ko.ts +++ b/gui/src/i18n/ko.ts @@ -426,6 +426,23 @@ export const ko: Record = { "compactionRouting.loadFailed": "압축 설정을 불러올 수 없습니다.", "compactionRouting.saved": "압축 설정을 저장했습니다.", "compactionRouting.saveFailed": "저장하지 못했습니다. 변경 사항은 유지됩니다. 다시 시도하세요.", + "memoryModels.title": "메모리 라우팅", + "memoryModels.description": "Codex는 세션이 끝난 뒤 백그라운드에서 메모리를 작성합니다. 각 단계에 쓸 모델을 고르거나 Codex의 기본 선택을 그대로 두세요.", + "memoryModels.infoLabel": "Extract와 Consolidation이 무엇인가요?", + "memoryModels.info": "Codex는 끝난 세션을 두 단계로 메모리로 만듭니다. Extract는 끝난 세션 하나를 읽고 무슨 일이 있었는지 적습니다. 세션마다 메모 한 장, 즉 작은 호출이 많습니다. Consolidation은 그 메모를 모아 다음 세션 시작에 Codex가 읽는 메모리 파일에 씁니다. 드물게 실행되지만 파일을 수정합니다. 각 단계가 자기 모델을 요청하기 때문에 여기에 따로 표시됩니다.", + "memoryModels.extract": "추출", + "memoryModels.extractHint": "끝난 세션마다 원시 메모리로 요약합니다. 세션당 한 번 실행됩니다.", + "memoryModels.consolidation": "통합", + "memoryModels.consolidationHint": "원시 메모리를 Codex가 나중에 읽는 메모리 파일로 합칩니다. 드물게 실행되며 파일을 수정합니다.", + "memoryModels.model": "모델", + "memoryModels.effort": "추론 노력", + "memoryModels.off": "사용 안 함 — Codex 기본값", + "memoryModels.defaultEffort": "Codex 기본값", + "memoryModels.dataNotice": "선택한 모델은 Codex가 메모리용으로 요약하는 세션 텍스트를 받습니다.", + "memoryModels.accountNotice": "여기서 선택하지 않으면 메모리 요청은 다른 네이티브 모델과 마찬가지로 OpenAI 계정으로 갑니다.", + "memoryModels.loadFailed": "메모리 설정을 불러오지 못했습니다.", + "memoryModels.saved": "메모리 설정을 저장했습니다.", + "memoryModels.saveFailed": "저장하지 못했습니다. 변경 사항은 그대로 있습니다. 다시 시도하세요.", "dash.shadowCallIntercept": "쉐도우 호출 가로채기", "dash.shadowCallInterceptHint": "Codex 앱이 제목·커밋 메시지 생성에 쓰는 백그라운드 호출({models})을 가로채 선택한 모델로 바꿉니다.", "dash.shadowCallWarning": "⚠ 활성화하면 {models} 요청이 모두 선택한 모델로 대체됩니다.", diff --git a/gui/src/i18n/ru.ts b/gui/src/i18n/ru.ts index 16a41bf6975..bdeb4e64998 100644 --- a/gui/src/i18n/ru.ts +++ b/gui/src/i18n/ru.ts @@ -431,6 +431,23 @@ export const ru: Record = { "compactionRouting.loadFailed": "Не удалось загрузить настройки сжатия.", "compactionRouting.saved": "Настройки сжатия сохранены.", "compactionRouting.saveFailed": "Не удалось сохранить. Изменения остались; попробуйте снова.", + "memoryModels.title": "Маршрутизация памяти", + "memoryModels.description": "Codex пишет память в фоне после завершения сессии. Выберите модель для каждого шага или оставьте выбор Codex.", + "memoryModels.infoLabel": "Что такое Extract и Consolidation?", + "memoryModels.info": "Codex превращает завершённые сессии в память за два шага. Extract читает одну завершённую сессию и записывает, что в ней произошло: одна заметка на сессию, то есть много небольших вызовов. Consolidation берёт эти заметки и записывает их в файлы памяти, которые Codex читает в начале следующих сессий. Запускается редко, но изменяет файлы. Каждый шаг сам запрашивает модель, поэтому они показаны отдельно.", + "memoryModels.extract": "Извлечение", + "memoryModels.extractHint": "Сводит каждую завершённую сессию в одну сырую запись. Запускается раз на сессию.", + "memoryModels.consolidation": "Консолидация", + "memoryModels.consolidationHint": "Сводит сырые записи в файлы памяти, которые Codex читает позже. Запускается редко и изменяет файлы.", + "memoryModels.model": "Модель", + "memoryModels.effort": "Усилие рассуждений", + "memoryModels.off": "Выключено — по умолчанию Codex", + "memoryModels.defaultEffort": "Как в Codex", + "memoryModels.dataNotice": "Выбранная модель получает текст сессии, который Codex сжимает для памяти.", + "memoryModels.accountNotice": "Без выбора здесь запросы памяти идут в ваш аккаунт OpenAI, как любая другая нативная модель.", + "memoryModels.loadFailed": "Не удалось загрузить настройки памяти.", + "memoryModels.saved": "Настройки памяти сохранены.", + "memoryModels.saveFailed": "Не удалось сохранить. Ваши изменения на месте; попробуйте снова.", "dash.shadowCallIntercept": "Перехват теневых вызовов", "dash.shadowCallInterceptHint": "Перехватывает фоновые служебные вызовы Codex App ({models}: генерация заголовков, сообщений коммитов) и перенаправляет их на выбранную вами модель.", "dash.shadowCallWarning": "⚠ Когда функция включена, ВСЕ запросы к {models} будут заменены выбранной моделью.", diff --git a/gui/src/i18n/tr.ts b/gui/src/i18n/tr.ts index b592e84752c..ac5097108fa 100644 --- a/gui/src/i18n/tr.ts +++ b/gui/src/i18n/tr.ts @@ -432,6 +432,23 @@ export const tr: Record = { "compactionRouting.loadFailed": "Özetleme ayarları yüklenemedi.", "compactionRouting.saved": "Özetleme ayarları kaydedildi.", "compactionRouting.saveFailed": "Kaydedilemedi. Değişiklikleriniz korunuyor; tekrar deneyin.", + "memoryModels.title": "Bellek yönlendirmesi", + "memoryModels.description": "Codex, oturum bittikten sonra belleği arka planda yazar. Her adım için modeli siz seçin ya da Codex'in kendi seçiminde bırakın.", + "memoryModels.infoLabel": "Extract ve Consolidation nedir?", + "memoryModels.info": "Codex, biten oturumları iki adımda belleğe dönüştürür. Extract biten bir oturumu okuyup ne olduğunu not eder: oturum başına bir not, yani çok sayıda küçük çağrı. Consolidation bu notları alıp Codex'in sonraki oturumların başında okuduğu bellek dosyalarına yazar. Seyrek çalışır ama dosyaları düzenler. Her adım kendi modelini ister, bu yüzden burada ayrı görünürler.", + "memoryModels.extract": "Çıkarım", + "memoryModels.extractHint": "Her biten oturumu bir ham belleğe özetler. Oturum başına bir kez çalışır.", + "memoryModels.consolidation": "Birleştirme", + "memoryModels.consolidationHint": "Ham bellekleri Codex'in sonra okuduğu bellek dosyalarında birleştirir. Seyrek çalışır ve dosyaları düzenler.", + "memoryModels.model": "Model", + "memoryModels.effort": "Akıl yürütme çabası", + "memoryModels.off": "Kapalı — Codex varsayılanı", + "memoryModels.defaultEffort": "Codex varsayılanı", + "memoryModels.dataNotice": "Seçilen model, Codex'in bellek için özetlediği oturum metnini alır.", + "memoryModels.accountNotice": "Burada seçim yapılmazsa bellek istekleri, diğer yerel modeller gibi OpenAI hesabınıza gider.", + "memoryModels.loadFailed": "Bellek ayarları yüklenemedi.", + "memoryModels.saved": "Bellek ayarları kaydedildi.", + "memoryModels.saveFailed": "Kaydedilemedi. Değişiklikleriniz duruyor; tekrar deneyin.", "dash.shadowCallIntercept": "Gölge Çağrı Yakalama", "dash.shadowCallInterceptHint": "Codex App'in arka plan yardımcı çağrılarını ({models}) başlık oluşturma ve commit mesajları için yakalar ve seçtiğiniz modele yönlendirir.", "dash.shadowCallWarning": "⚠ Etkinleştirildiğinde, {models} için olan TÜM istekler seçilen modelle değiştirilecektir.", diff --git a/gui/src/i18n/vi.ts b/gui/src/i18n/vi.ts index 4be42b9ebc6..a62f08aa54d 100644 --- a/gui/src/i18n/vi.ts +++ b/gui/src/i18n/vi.ts @@ -430,6 +430,23 @@ export const vi: Record = { "compactionRouting.loadFailed": "Không thể tải cài đặt nén.", "compactionRouting.saved": "Đã lưu cài đặt nén.", "compactionRouting.saveFailed": "Không thể lưu. Thay đổi của bạn vẫn còn; hãy thử lại.", + "memoryModels.title": "Định tuyến bộ nhớ", + "memoryModels.description": "Codex ghi bộ nhớ ở chế độ nền sau khi phiên kết thúc. Chọn mô hình cho từng bước, hoặc để Codex tự chọn.", + "memoryModels.infoLabel": "Extract và Consolidation là gì?", + "memoryModels.info": "Codex biến các phiên đã kết thúc thành bộ nhớ qua hai bước. Extract đọc một phiên đã kết thúc và ghi lại những gì đã diễn ra: một ghi chú cho mỗi phiên, nên có nhiều lời gọi nhỏ. Consolidation lấy các ghi chú đó và ghi vào các tệp bộ nhớ mà Codex đọc khi bắt đầu các phiên sau. Hiếm khi chạy nhưng có sửa tệp. Mỗi bước tự yêu cầu mô hình riêng, nên ở đây chúng được tách riêng.", + "memoryModels.extract": "Trích xuất", + "memoryModels.extractHint": "Tóm tắt mỗi phiên đã kết thúc thành một bộ nhớ thô. Chạy một lần mỗi phiên.", + "memoryModels.consolidation": "Hợp nhất", + "memoryModels.consolidationHint": "Hợp nhất các bộ nhớ thô vào tệp bộ nhớ mà Codex đọc sau này. Hiếm khi chạy và có sửa tệp.", + "memoryModels.model": "Mô hình", + "memoryModels.effort": "Mức suy luận", + "memoryModels.off": "Tắt — mặc định của Codex", + "memoryModels.defaultEffort": "Mặc định của Codex", + "memoryModels.dataNotice": "Mô hình được chọn sẽ nhận nội dung phiên mà Codex tóm tắt cho bộ nhớ.", + "memoryModels.accountNotice": "Nếu không chọn ở đây, các yêu cầu bộ nhớ sẽ đi tới tài khoản OpenAI của bạn như mọi mô hình gốc khác.", + "memoryModels.loadFailed": "Không tải được cài đặt bộ nhớ.", + "memoryModels.saved": "Đã lưu cài đặt bộ nhớ.", + "memoryModels.saveFailed": "Không lưu được. Thay đổi của bạn vẫn còn; hãy thử lại.", "dash.shadowCallIntercept": "Shadow Call Intercept", "dash.shadowCallInterceptHint": "Chặn các lệnh gọi helper nền ({models}) của ứng dụng Codex để tạo tiêu đề và commit messages, sau đó chuyển hướng chúng đến model bạn đã chọn.", "dash.shadowCallWarning": "⚠ Khi được bật, TẤT CẢ yêu cầu đối với {models} sẽ được thay thế bằng model được chọn.", diff --git a/gui/src/i18n/zh-TW.ts b/gui/src/i18n/zh-TW.ts index 9285fe2a451..86c850887ff 100644 --- a/gui/src/i18n/zh-TW.ts +++ b/gui/src/i18n/zh-TW.ts @@ -308,6 +308,23 @@ export const zhTW: Record = { "compactionRouting.loadFailed": "無法載入壓縮設定。", "compactionRouting.saved": "壓縮設定已儲存。", "compactionRouting.saveFailed": "儲存失敗。變更仍然保留,請重試。", + "memoryModels.title": "記憶路由", + "memoryModels.description": "工作階段結束後,Codex 會在背景寫入記憶。為每個步驟選擇模型,或保留 Codex 自己的選擇。", + "memoryModels.infoLabel": "Extract 和 Consolidation 是什麼?", + "memoryModels.info": "Codex 用兩個步驟把結束的工作階段變成記憶。Extract 讀取一個已結束的工作階段並記下其中發生的事:每個工作階段一張筆記,因此會有很多小型請求。Consolidation 把這些筆記寫進 Codex 在後續工作階段開始時讀取的記憶檔案。很少執行,但會修改檔案。兩個步驟各自要求自己的模型,所以這裡分開顯示。", + "memoryModels.extract": "擷取", + "memoryModels.extractHint": "把每個結束的工作階段彙整成一筆原始記憶。每個工作階段執行一次。", + "memoryModels.consolidation": "合併", + "memoryModels.consolidationHint": "把原始記憶合併進 Codex 之後讀取的記憶檔案。很少執行,而且會修改檔案。", + "memoryModels.model": "模型", + "memoryModels.effort": "推理強度", + "memoryModels.off": "關閉 — Codex 預設", + "memoryModels.defaultEffort": "Codex 預設", + "memoryModels.dataNotice": "所選模型會收到 Codex 用於記憶的工作階段文字。", + "memoryModels.accountNotice": "這裡不選,記憶請求會像其他原生模型一樣送往你的 OpenAI 帳戶。", + "memoryModels.loadFailed": "無法載入記憶設定。", + "memoryModels.saved": "記憶設定已儲存。", + "memoryModels.saveFailed": "儲存失敗。你的變更還在,請再試一次。", "dash.shadowCallIntercept": "影子呼叫攔截", "dash.shadowCallInterceptHint": "攔截 Codex 應用的背景 helper 呼叫({models})以生成標題與提交訊息,並將它們重定向到您選擇的模型。", "dash.shadowCallWarning": "⚠ 啟用後,{models} 的所有請求將被替換為所選模型。", diff --git a/gui/src/i18n/zh.ts b/gui/src/i18n/zh.ts index d30a30fe4e1..73b35966385 100644 --- a/gui/src/i18n/zh.ts +++ b/gui/src/i18n/zh.ts @@ -426,6 +426,23 @@ export const zh: Record = { "compactionRouting.loadFailed": "无法加载压缩设置。", "compactionRouting.saved": "压缩设置已保存。", "compactionRouting.saveFailed": "保存失败。更改仍然保留,请重试。", + "memoryModels.title": "记忆路由", + "memoryModels.description": "会话结束后,Codex 会在后台写入记忆。为每个步骤选择模型,或保留 Codex 自己的选择。", + "memoryModels.infoLabel": "Extract 和 Consolidation 是什么?", + "memoryModels.info": "Codex 分两步把结束的会话变成记忆。Extract 读取一个已结束的会话并记下其中的内容:每个会话一张笔记,因此有很多小请求。Consolidation 把这些笔记写进 Codex 在后续会话开始时读取的记忆文件。很少运行,但会修改文件。两步各自请求自己的模型,所以这里分开展示。", + "memoryModels.extract": "提取", + "memoryModels.extractHint": "把每个结束的会话汇总成一条原始记忆。每个会话运行一次。", + "memoryModels.consolidation": "整合", + "memoryModels.consolidationHint": "把原始记忆合并进 Codex 之后读取的记忆文件。很少运行,并会修改文件。", + "memoryModels.model": "模型", + "memoryModels.effort": "推理强度", + "memoryModels.off": "关闭 — Codex 默认", + "memoryModels.defaultEffort": "Codex 默认", + "memoryModels.dataNotice": "所选模型会收到 Codex 用于记忆的会话文本。", + "memoryModels.accountNotice": "这里不选,记忆请求会像其他原生模型一样发往你的 OpenAI 账户。", + "memoryModels.loadFailed": "无法加载记忆设置。", + "memoryModels.saved": "记忆设置已保存。", + "memoryModels.saveFailed": "保存失败。你的改动仍在,请重试。", "dash.shadowCallIntercept": "影子调用拦截", "dash.shadowCallInterceptHint": "拦截 Codex 应用的后台辅助调用({models}:标题生成、提交消息)并重定向到所选模型。", "dash.shadowCallWarning": "⚠ 启用后,所有对 {models} 的请求都将被替换为所选模型。", diff --git a/gui/src/pages/dashboard-overview-panels.tsx b/gui/src/pages/dashboard-overview-panels.tsx index 9d6a9e43684..afe2874d380 100644 --- a/gui/src/pages/dashboard-overview-panels.tsx +++ b/gui/src/pages/dashboard-overview-panels.tsx @@ -1,5 +1,6 @@ import CompactionRoutingPanel from "../components/CompactionRoutingPanel"; import MemoryObservabilityCard from "../components/MemoryObservabilityCard"; +import MemoryModelsPanel from "../components/MemoryModelsPanel"; import type { useDashboardData } from "./use-dashboard-data"; import { DashboardEffortCapPanel, @@ -20,6 +21,7 @@ export function DashboardOverviewPanels(props: Dash) { + ); diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index f3f16fae58b..6e8323c28b6 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -200,6 +200,7 @@ "start-args.test.ts": "cli", "start-ownership-publication.test.ts": "cli", "responses-core-modules.test.ts": "responses", + "responses-memory-models.test.ts": "responses", "responses-grok-devin-preflight.test.ts": "responses", "responses-passthrough-transient-policy.test.ts": "responses", "responses-spend-ledger-wiring.test.ts": "responses", diff --git a/src/config/diagnostics.ts b/src/config/diagnostics.ts index d1ab81747c4..4c630cf169f 100644 --- a/src/config/diagnostics.ts +++ b/src/config/diagnostics.ts @@ -64,6 +64,7 @@ import { spendSchema, compactionRoutingSchema, skillsConfigSchema, + memoryModelsSchema, } from "./schema/leaf-validators"; export type ConfigDiagnostics = { @@ -611,6 +612,10 @@ export function validateConfigCandidate(value: unknown): { ok: true; config: Ocx if (compactionRouting !== undefined && !compactionRoutingSchema.safeParse(compactionRouting).success) { return { ok: false, error: "schema_invalid: compactionRouting: requires a nonblank model, an optional valid reasoningEffort, and optional non-repeating triggers drawn from \"manual\" and \"auto\"" }; } + const memoryModels = rawConfigRecord(value)?.memoryModels; + if (memoryModels !== undefined && !memoryModelsSchema.safeParse(memoryModels).success) { + return { ok: false, error: "schema_invalid: memoryModels: requires a nonblank model and an optional declared reasoningEffort per configured phase, and no other fields" }; + } const boundaryError = compactionRecoveryConfigError(value) ?? configReasoningPinsConfigError(value) ?? blankHostnameError(value) ?? claudeSubagentEffortError(value) diff --git a/src/config/load-degrade.ts b/src/config/load-degrade.ts index aa8cc5e94c4..1069a6ef2de 100644 --- a/src/config/load-degrade.ts +++ b/src/config/load-degrade.ts @@ -122,6 +122,18 @@ export function warnDegradedTopLevelOptIns(rawParsed: unknown, validated: OcxCon if (compactionRecoveryConfigError(rawParsed)) console.warn("⚠️ invalid compactionRecovery disabled; the original compaction failure is preserved"); warnDegradedStreamMode(rawParsed, validated); warnDegradedCompactionRouting(rawParsed, validated); + warnDegradedMemoryModels(rawParsed, validated); +} + +/** + * A malformed `memoryModels` entry disables that phase rather than failing the whole schema, so + * say so once: silently keeping the native model is the outcome a typo must not produce quietly. + */ +export function warnDegradedMemoryModels(rawParsed: unknown, validated: OcxConfig): void { + if (!rawParsed || typeof rawParsed !== "object") return; + const raw = (rawParsed as Record).memoryModels; + if (raw === undefined || validated.memoryModels !== undefined) return; + console.warn("\u26a0\ufe0f config.json memoryModels is invalid (expected { extract?: { model, reasoningEffort? }, consolidation?: { model, reasoningEffort? } } with a nonblank model and a declared effort per phase) \u2014 Codex keeps its own model for the memory pipeline"); } /** diff --git a/src/config/schema/config-schema.ts b/src/config/schema/config-schema.ts index a855202a29a..9ec8ff54c91 100644 --- a/src/config/schema/config-schema.ts +++ b/src/config/schema/config-schema.ts @@ -25,6 +25,7 @@ import { codexAccountNamespacesSchema, modelPinnedEffortsSchema, compactionRoutingSchema, + memoryModelsSchema, modelPreferHostedToolsConfigError, providerModelCostsConfigError, providerRelativeSendPathConfigError, @@ -160,6 +161,9 @@ export const configSchema = z.object({ modelPinnedEfforts: modelPinnedEffortsSchema.optional(), compactionRouting: compactionRoutingSchema.optional().catch(undefined), compactionRecovery: compactionRecoverySchema.optional().catch(undefined), + // A hand-edited malformed phase disables that phase instead of rejecting providers/apiKeys; + // the management write boundary (validateConfigCandidate) still refuses the bad value. + memoryModels: memoryModelsSchema.optional().catch(undefined), defaultProvider: z.string().min(1).default("openai"), defaultModelAliases: z.boolean().optional(), // Malformed hand edits disable this opt-in projection without rejecting providers. diff --git a/src/config/schema/leaf-validators.ts b/src/config/schema/leaf-validators.ts index 8de6060f5b2..e7d06cff4f7 100644 --- a/src/config/schema/leaf-validators.ts +++ b/src/config/schema/leaf-validators.ts @@ -55,6 +55,21 @@ export const compactionRoutingSchema = z.object({ .optional(), }).strict(); +/** + * One phase of Codex's memory pipeline. A present phase must name a model: the GUI's "Off" + * removes the phase instead of blanking it, so an empty entry would only ever come from a + * hand-edited file, where failing the write is the honest answer. + */ +const memoryModelSettingSchema = z.object({ + model: z.string().trim().min(1), + reasoningEffort: z.string().refine(value => pinnedReasoningEffortConfigError(value) === null).optional(), +}).strict(); + +export const memoryModelsSchema = z.object({ + extract: memoryModelSettingSchema.optional(), + consolidation: memoryModelSettingSchema.optional(), +}).strict(); + /** * Bounds for the opt-in same-target 429 wait-and-retry policy. Single source of truth * shared by the config schema, the load-time sanitizer, and the management write diff --git a/src/server/management/config-routes.ts b/src/server/management/config-routes.ts index 45cfee41f76..73b89558d57 100644 --- a/src/server/management/config-routes.ts +++ b/src/server/management/config-routes.ts @@ -1,4 +1,4 @@ -import { compactionRoutingSchema } from "../../config/schema/leaf-validators"; +import { compactionRoutingSchema, memoryModelsSchema } from "../../config/schema/leaf-validators"; import { compactionRecoverySchema } from "../../config/schema/compaction-recovery"; import { captureConfigTopLevelRollback } from "../../config/rebase-provenance"; import type { IntegrationClientId } from "../../integrations/registry"; @@ -374,6 +374,8 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise${applied.to}` : applied.to; + if (isInjectionDebugEnabled()) { + injectionDebugLog(`[opencodex] ${route.modelId}: memory ${phase} effort applied (${applied.from ?? "none"} -> ${applied.to})`); + } + } + } + } + { const { applyEffortCap, effortCapAppliesTo, supportedLadderFor } = await import("../effort-policy"); const surface = collabSurface(parsed); diff --git a/src/server/responses/core-options.ts b/src/server/responses/core-options.ts index a09bf122941..aea7f850800 100644 --- a/src/server/responses/core-options.ts +++ b/src/server/responses/core-options.ts @@ -145,6 +145,8 @@ export interface HandleResponsesOptions { comboAttempt?: boolean; /** Internal handoff: this combo was selected by shadow-call interception. */ shadowCallIntercepted?: boolean; + /** Internal handoff: the memory phase this turn belongs to, so combo children keep its routing. */ + memoryModelPhase?: "extract" | "consolidation"; compactionRoutingOverride?: CompactionRoutingOverride | null; /** Internal combo handoff for one parent-validated continuation snapshot. */ comboReplaySnapshot?: { diff --git a/src/server/responses/memory-models.ts b/src/server/responses/memory-models.ts new file mode 100644 index 00000000000..f54a0f79593 --- /dev/null +++ b/src/server/responses/memory-models.ts @@ -0,0 +1,158 @@ +/** + * Model routing for Codex's own memory pipeline. + * + * Codex writes memories in two background phases, and both ask the provider for a bare native + * model: Phase 1 ("extract") summarizes one finished thread per call and asks for + * `gpt-5.6-luna` at effort `low`; Phase 2 ("consolidation") is one agent run that merges those + * summaries into the files under `$CODEX_HOME/memories` and asks for `gpt-5.6-terra` at effort + * `medium`. Without a configured target both resolve through the canonical OpenAI route even when + * every ordinary turn is routed elsewhere — and Phase 1 additionally looks like the app's + * title/commit helper traffic, because the app uses the same model id for those. + * + * A phase is therefore recognized from Codex's own turn metadata, never inferred from the model + * id, the timing, or the token counts. Phase 1 sends `request_kind: "memory"`; both phases carry + * `thread_source: "memory_consolidation"`, and Phase 2 additionally arrives with + * `x-openai-subagent: memory_consolidation` (codex-rs `core/src/responses_metadata.rs`). + */ +import type { OcxConfig, OcxParsedRequest } from "../../types"; +import { isDeclaredReasoningEffort } from "../../reasoning-effort"; + +/** The two phases Codex runs, in the order it runs them. */ +export type MemoryModelPhase = "extract" | "consolidation"; + +/** codex-rs serializes both keys below into the JSON `x-codex-turn-metadata` header. */ +const TURN_METADATA_HEADER = "x-codex-turn-metadata"; +const REQUEST_KIND_KEY = "request_kind"; +const THREAD_SOURCE_KEY = "thread_source"; +/** `CodexResponsesRequestKind::Memory` (codex-rs `core/src/responses_metadata.rs`). */ +const MEMORY_REQUEST_KIND = "memory"; +/** `ThreadSource::MemoryConsolidation` / `InternalSessionSource::MemoryConsolidation`. */ +const MEMORY_THREAD_SOURCE = "memory_consolidation"; +const SUBAGENT_HEADER = "x-openai-subagent"; + +function record(value: unknown): Record | undefined { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? value as Record + : undefined; +} + +/** One metadata copy's verdict. `"none"` is a well-formed copy that is not a memory turn. */ +type CopyVerdict = MemoryModelPhase | "none"; + +function verdictOf(parsed: Record): CopyVerdict { + // Phase 1's detached request names the memory kind explicitly. Phase 2 is an ordinary turn + // inside the `memory_consolidation` thread, so its thread source is the only signal there. + if (parsed[REQUEST_KIND_KEY] === MEMORY_REQUEST_KIND) return "extract"; + if (parsed[THREAD_SOURCE_KEY] === MEMORY_THREAD_SOURCE) return "consolidation"; + return "none"; +} + +/** + * Recognize a memory-pipeline turn, or null. + * + * Every copy of the turn metadata the request carries must agree — the same rule + * `applyCompactionRoutingOverride` applies to compaction turns: a request that contradicts itself + * is not a memory turn, so neither copy can widen what the setting covers. The sub-agent header is + * accepted on its own because the WS bridge rebuilds internal requests from a header allowlist and + * may deliver only that copy. + */ +export function detectMemoryModelPhase( + body: unknown, + headers: Headers, + options: { transport?: "websocket" } = {}, +): MemoryModelPhase | null { + const metadata: unknown[] = []; + const header = headers.get(TURN_METADATA_HEADER); + if (options.transport !== "websocket" && header !== null) metadata.push(header); + const client = record(record(body)?.["client_metadata"]); + if (client && Object.hasOwn(client, TURN_METADATA_HEADER)) metadata.push(client[TURN_METADATA_HEADER]); + + let verdict: CopyVerdict | null = null; + for (const value of metadata) { + if (typeof value !== "string") return null; + let parsed: Record | undefined; + try { + parsed = record(JSON.parse(value)); + } catch { + return null; + } + if (!parsed) return null; + const copy = verdictOf(parsed); + if (verdict !== null && verdict !== copy) return null; + verdict = copy; + } + if (verdict === "extract" || verdict === "consolidation") return verdict; + return headers.get(SUBAGENT_HEADER) === MEMORY_THREAD_SOURCE ? "consolidation" : null; +} + +/** The configured destination for one phase, or undefined while the phase keeps Codex's choice. */ +export function configuredMemoryModel( + config: Pick | undefined, + phase: MemoryModelPhase, +): { model: string; reasoningEffort?: string } | undefined { + const setting = config?.memoryModels?.[phase]; + if (!setting) return undefined; + const model = typeof setting.model === "string" ? setting.model.trim() : ""; + if (!model) return undefined; + const effort = typeof setting.reasoningEffort === "string" ? setting.reasoningEffort : undefined; + return { model, ...(effort ? { reasoningEffort: effort } : {}) }; +} + +/** + * Force the configured effort onto a memory turn. + * + * Codex hard-codes the phase effort (`low` for Phase 1, `medium` for Phase 2) and has no config + * key for it, so this is the only place the operator's choice can land. Both wire shapes are + * written: `parsed.options.reasoning` feeds the routed adapters, `_rawBody.reasoning.effort` feeds + * the ChatGPT passthrough serializer — the same dual-shape contract `applyPinnedEffort` uses. + */ +export function applyMemoryModelEffort( + parsed: OcxParsedRequest, + config: Pick | undefined, + phase: MemoryModelPhase, +): { from: string | undefined; to: string } | null { + const effort = configuredMemoryModel(config, phase)?.reasoningEffort; + if (!effort || !isDeclaredReasoningEffort(effort)) return null; + const requested = parsed.options.reasoning; + if (requested === effort) return null; + parsed.options.reasoning = effort; + const raw = parsed._rawBody as { reasoning?: { effort?: string } } | undefined; + if (raw && typeof raw === "object") { + raw.reasoning = { ...(record(raw.reasoning) ?? {}), effort } as { effort?: string }; + } + return { from: requested, to: effort }; +} + +/** Route reason recorded for a routed memory turn, so the request log names the phase. */ +export function memoryModelRouteReason(phase: MemoryModelPhase): string { + return phase === "extract" ? "memory-extract" : "memory-consolidation"; +} + +/** Non-retryable: the target stays unavailable until the operator changes the setting. */ +export const MEMORY_MODEL_TARGET_UNAVAILABLE_CODE = "memory_model_target_unavailable"; +export const MEMORY_MODEL_TARGET_UNAVAILABLE_STATUS = 409; + +const warnedTargets = new Set(); + +/** + * A configured phase destination that stopped resolving fails its call once, clearly, instead of + * silently falling back to the native model the operator routed away from — the same contract the + * shadow intercept uses for its single target. + */ +export function memoryModelTargetUnavailableResponse( + phase: MemoryModelPhase, + model: string, + detail: string, +): Response { + const message = `Memory ${phase} model "${model}" is unavailable: ${detail}. ` + + "Choose another model in the Memory Models settings or re-enable its provider."; + const key = `${phase}\u0000${model}\u0000${detail}`; + if (!warnedTargets.has(key)) { + warnedTargets.add(key); + console.warn(`memory-models: ${message}`); + } + return new Response( + JSON.stringify({ error: { message, type: "invalid_request_error", code: MEMORY_MODEL_TARGET_UNAVAILABLE_CODE } }), + { status: MEMORY_MODEL_TARGET_UNAVAILABLE_STATUS, headers: { "Content-Type": "application/json" } }, + ); +} diff --git a/src/server/responses/request-prepare.ts b/src/server/responses/request-prepare.ts index 2ac1679a72f..8d1c31b9197 100644 --- a/src/server/responses/request-prepare.ts +++ b/src/server/responses/request-prepare.ts @@ -13,7 +13,14 @@ import { } from "./core-errors"; import { parseSyntheticRowId } from "../fast-row"; import { resolveComboId, comboIdFromRawBody, NoAvailableComboTargetsError } from "../../combos"; -import { INTERCEPT_TARGET_UNAVAILABLE_CODE, interceptTargetUnavailableResponse, resolveShadowCallTarget } from "./shadow-target-availability"; +import { INTERCEPT_TARGET_UNAVAILABLE_CODE, interceptTargetUnavailableResponse, resolveChosenTarget, resolveShadowCallTarget } from "./shadow-target-availability"; +import { + MEMORY_MODEL_TARGET_UNAVAILABLE_CODE, + configuredMemoryModel, + detectMemoryModelPhase, + memoryModelRouteReason, + memoryModelTargetUnavailableResponse, +} from "./memory-models"; import { recallComboForLane } from "./combo-session-recall"; import { sessionLaneIdFromRequest, @@ -175,6 +182,19 @@ export async function prepareResponsesRequest( transport: options.inboundTransport, }); } + // Codex's memory pipeline names a destination per phase. The phase is read from Codex's own turn + // metadata, never from the model id: Phase 1 shares `gpt-5.6-luna` with the app's title/commit + // helper calls. Read here, ahead of the shadow intercept below, because the phase decision is the + // more specific of the two settings and must be the one that survives when both match one request. + const memoryModelPhase = options.memoryModelPhase + ?? (!options.comboAttempt && !options.compactionRoutingOverride && inboundWire === "responses" + ? detectMemoryModelPhase(body, req.headers, { transport: options.inboundTransport }) ?? undefined + : undefined); + const memoryModelTarget = memoryModelPhase ? configuredMemoryModel(config, memoryModelPhase) : undefined; + // A combo child is a synthetic replay of the parent's decision: its model is already the target's + // concrete provider/model, so neither site below may rewrite or re-resolve it. It keeps the phase + // through `options.memoryModelPhase` instead, which is what applies the phase effort. + const memoryModelApplies = memoryModelTarget !== undefined && options.comboAttempt !== true; options.onRequestBodyParsed?.(body); // An effort row naming a table-less combo (`combo/x--high`) must reach the combo dispatcher // as its base id, so the selector is normalized here, before comboIdFromRawBody reads model. @@ -229,11 +249,22 @@ export async function prepareResponsesRequest( // hops — which only exist inside that loop — are unreachable (#4129). Rewrite the selector // here instead, before comboIdFromRawBody reads `model`, and identify the combo by CONFIG // LOOKUP so the check can never observe a one-candidate collapse. + // A memory target that names a combo has to reach the combo dispatcher as `model`, or its own + // failover loop is unreachable (#4129) — the same reason the shadow intercept rewrites its combo + // target here. Every other target is resolved at the late site, where the admission scope exists. + let memoryModelComboRouted = false; + if (memoryModelApplies && memoryModelTarget && body && typeof body === "object" && !Array.isArray(body)) { + const memoryComboId = resolveComboId(config, memoryModelTarget.model); + if (memoryComboId && Object.hasOwn(config.combos ?? {}, memoryComboId)) { + memoryModelComboRouted = true; + (body as Record).model = memoryModelTarget.model; + } + } let shadowCallIntercepted = false; // A spawned sub-agent turn names its model on purpose; gpt-6-luna is both the helper // slug and a default sub-agent model, so neither intercept site may rewrite that turn. const threadSpawn = isThreadSpawnRequest(req.headers); - if (!options.comboAttempt && !options.compactionRoutingOverride && !threadSpawn && body && typeof body === "object" && !Array.isArray(body)) { + if (!options.comboAttempt && !options.compactionRoutingOverride && !threadSpawn && !memoryModelApplies && body && typeof body === "object" && !Array.isArray(body)) { const shadowIntercept = config.shadowCallIntercept; const rawShadowModel = (body as { model?: unknown }).model; if (shadowIntercept?.enabled && shadowIntercept.model && typeof rawShadowModel === "string" @@ -259,6 +290,9 @@ export async function prepareResponsesRequest( // Concrete combo child selectors no longer match the shadow source model. Carry the // interception decision explicitly so provider-specific helper isolation still applies. shadowCallIntercepted, + // Same handoff for a memory phase whose target is a combo: the child keeps the phase's effort + // override and stays out of the parent conversation. + memoryModelPhase: memoryModelComboRouted ? memoryModelPhase : undefined, // The original request body was accepted above. Combo children are synthetic // replays and must not repeat the caller-owned timeout transition. onRequestBodyRead: undefined, @@ -391,6 +425,10 @@ export async function prepareResponsesRequest( } if (cursorClientThreadId) parsed._cursorClientThreadId = cursorClientThreadId; if (options.shadowCallIntercepted === true) parsed._cursorIsolateConversation = true; + if (options.memoryModelPhase !== undefined) { + parsed._memoryModelPhase = options.memoryModelPhase; + parsed._cursorIsolateConversation = true; + } } catch (err) { if (isTranslatorBudgetExceededError(err)) { return formatErrorResponse(413, "request_too_large", "request translation buffer exceeded the safe limit", { @@ -494,9 +532,28 @@ export async function prepareResponsesRequest( : parsed._compactionRequest === true ? routeCompactionModel(config, modelId, evidenceFromBody(parsed._rawBody)) : routeModel(config, modelId, evidenceFromBody(parsed._rawBody))); + // The phase's destination. Resolved through the admission-scoped resolver every other route + // uses, and it fails closed exactly like the shadow target: falling back to the native model + // would spend the quota the operator routed away from, without their choosing it. + let memoryRoute: RouteResult | undefined; + if (memoryModelApplies && memoryModelPhase && memoryModelTarget) { + const memoryTarget = resolveChosenTarget(memoryModelTarget.model, resolveRoute); + if ("unavailable" in memoryTarget) { + logCtx.errorCode = MEMORY_MODEL_TARGET_UNAVAILABLE_CODE; + return memoryModelTargetUnavailableResponse(memoryModelPhase, memoryModelTarget.model, memoryTarget.unavailable); + } + credentialDomainWasRewritten = true; + parsed.modelId = memoryModelTarget.model; + if (parsed._rawBody && typeof parsed._rawBody === "object") { + (parsed._rawBody as { model?: string }).model = memoryModelTarget.model; + } + parsed._memoryModelPhase = memoryModelPhase; + parsed._cursorIsolateConversation = true; + memoryRoute = memoryTarget.route; + } const _sci = config.shadowCallIntercept; let shadowRoute: RouteResult | undefined; - if (!options.compactionRoutingOverride && !threadSpawn && _sci?.enabled && _sci.model && isShadowSourceModel(parsed.modelId, _sci.sourceModels)) { + if (!memoryRoute && !options.compactionRoutingOverride && !threadSpawn && _sci?.enabled && _sci.model && isShadowSourceModel(parsed.modelId, _sci.sourceModels)) { const sourcePrefix = shadowSourceModelPrefix(parsed.modelId, _sci.sourceModels)!; let sourceIdentity = { providerName: OPENAI_CODEX_PROVIDER_ID, modelId: sourcePrefix }; try { @@ -533,7 +590,11 @@ export async function prepareResponsesRequest( } } if (parsed._compactionRequest === true || options.compactionRoutingOverride) parsed._cursorIsolateConversation = true; - route = shadowRoute ?? resolveRoute(parsed.modelId); + route = memoryRoute ?? shadowRoute ?? resolveRoute(parsed.modelId); + // Name the phase in the persisted route decision, so the request log says why this turn went to + // the memory destination instead of leaving it looking like a plain user selection. Set here, on + // the resolved route, so a combo child's own route carries it too. + if (parsed._memoryModelPhase) route.routeReason = memoryModelRouteReason(parsed._memoryModelPhase); if (options.compactionRoutingOverride && !compactionRoutingKeepsProviderIdentity(config, options.compactionRoutingOverride, route)) { credentialDomainWasRewritten = true; // The destination does not share the conversation's credential domain, so it can neither diff --git a/src/server/responses/shadow-target-availability.ts b/src/server/responses/shadow-target-availability.ts index 814f99c9316..4c329fb2ff9 100644 --- a/src/server/responses/shadow-target-availability.ts +++ b/src/server/responses/shadow-target-availability.ts @@ -16,16 +16,21 @@ export const INTERCEPT_TARGET_UNAVAILABLE_CODE = "intercept_target_unavailable"; /** Non-retryable: the target stays unavailable until the operator changes the configuration. */ export const INTERCEPT_TARGET_UNAVAILABLE_STATUS = 409; -export type ShadowTargetResolution = { route: RouteResult } | { unavailable: string }; +export type ChosenTargetResolution = { route: RouteResult } | { unavailable: string }; +export type ShadowTargetResolution = ChosenTargetResolution; /** - * Resolve the configured target. Admission-scope refusals, exhausted combos and policy + * Resolve one operator-chosen target. Admission-scope refusals, exhausted combos and policy * evaluations keep their existing responses, so they are rethrown to the caller's handler. + * + * Shared by the shadow-call intercept and the memory-model routing: both name a single destination + * whose unavailability must not be papered over by the router's terminal default-provider + * fallback. `shadowCallTargetsIntersect` and the memory setting are the two callers. */ -export function resolveShadowCallTarget( +export function resolveChosenTarget( model: string, resolve: (model: string) => RouteResult, -): ShadowTargetResolution { +): ChosenTargetResolution { let route: RouteResult; try { route = resolve(model); @@ -44,6 +49,14 @@ export function resolveShadowCallTarget( return { route }; } +/** The shadow-call name for the shared resolver, kept because that is the surface's own vocabulary. */ +export function resolveShadowCallTarget( + model: string, + resolve: (model: string) => RouteResult, +): ShadowTargetResolution { + return resolveChosenTarget(model, resolve); +} + const warnedTargets = new Set(); export function interceptTargetUnavailableResponse(model: string, detail: string): Response { diff --git a/src/types/config.ts b/src/types/config.ts index 940aa2b06fd..e18679f4c07 100644 --- a/src/types/config.ts +++ b/src/types/config.ts @@ -738,6 +738,27 @@ export interface OcxConfig { }; /** Opt-in failure-only recovery; never replaces the initial compaction model. */ compactionRecovery?: { enabled: boolean; model: string; allowDevinInvalidArgument?: boolean }; + /** + * Destination model for Codex's own memory pipeline, per phase + * (src/server/responses/memory-models.ts). + * + * Codex runs Phase 1 ("extract") once per finished thread to summarize that thread's rollout, + * and Phase 2 ("consolidation") once as an agent run that merges the summaries into the files + * under `$CODEX_HOME/memories`. Both ask for a bare native model, so without an entry here they + * resolve through the canonical OpenAI route even when ordinary turns are routed elsewhere. + * + * A phase is recognized from Codex's turn metadata, never inferred from the model id, the timing + * or the token counts: Phase 1 shares `gpt-5.6-luna` with the app's title/commit helper calls, + * and `shadowCallIntercept` is the setting for those. A configured phase wins over that + * intercept, because the memory decision is the more specific one. + * + * `model` is required for a configured phase; omitting the phase (or its `model`) leaves Codex's + * own choice in place. `reasoningEffort` overrides the effort Codex hard-codes for that phase. + */ + memoryModels?: { + extract?: { model: string; reasoningEffort?: string }; + consolidation?: { model: string; reasoningEffort?: string }; + }; /** * Models hidden from Codex discovery without blocking direct proxy calls. Routed provider ids * are excluded from the catalog + /v1/models entirely. Account-qualified native ids hide only diff --git a/src/types/request.ts b/src/types/request.ts index 03faeb5afec..288de9383f4 100644 --- a/src/types/request.ts +++ b/src/types/request.ts @@ -143,6 +143,12 @@ export interface OcxParsedRequest { _compactionRequest?: boolean; /** Manual compaction moved to another provider: summarize portably even on a canonical ChatGPT target. */ _portableCompaction?: boolean; + /** + * Codex memory pipeline phase this turn belongs to, when `memoryModels` routes it + * (src/server/responses/memory-models.ts). Read at the effort choke point, which runs after the + * route is known. + */ + _memoryModelPhase?: "extract" | "consolidation"; /** * True when the current request newly introduced a stored compaction summary/marker. Historical * markers restored by previous_response_id expansion were already acknowledged and do not reset diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 01768b541df..7652e204f20 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -46,6 +46,7 @@ "start-args.test.ts": "cli", "start-ownership-publication.test.ts": "cli", "responses-core-modules.test.ts": "responses", + "responses-memory-models.test.ts": "responses", "responses-grok-devin-preflight.test.ts": "responses", "responses-passthrough-transient-policy.test.ts": "responses", "responses-spend-ledger-wiring.test.ts": "responses", diff --git a/tests/responses/responses-memory-models.test.ts b/tests/responses/responses-memory-models.test.ts new file mode 100644 index 00000000000..90f5a7cbae7 --- /dev/null +++ b/tests/responses/responses-memory-models.test.ts @@ -0,0 +1,304 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { + MEMORY_MODEL_TARGET_UNAVAILABLE_CODE, + applyMemoryModelEffort, + configuredMemoryModel, + detectMemoryModelPhase, +} from "../../src/server/responses/memory-models"; +import { handleResponses } from "../../src/server/responses"; +import { getDefaultConfig, validateConfigCandidate } from "../../src/config"; +import { configSchema } from "../../src/config/schema/config-schema"; +import { warnDegradedMemoryModels } from "../../src/config/load-degrade"; +import { clearComboSelectionState, clearComboTargetCooldowns } from "../../src/combos"; +import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; +import type { OcxConfig, OcxParsedRequest } from "../../src/types"; + +const originalFetch = globalThis.fetch; +/** The spend-journal writer lease is taken by startServer, so a bare handler call needs one. */ +let releaseSpendHome: (() => void) | undefined; + +/** Phase 1's shape: codex-rs marks the kind AND the thread source. */ +const extractMetadata = (extra: Record = {}) => + JSON.stringify({ request_kind: "memory", thread_source: "memory_consolidation", ...extra }); +/** Phase 2's shape: an ordinary turn inside the consolidation thread. */ +const consolidationMetadata = () => + JSON.stringify({ request_kind: "turn", thread_source: "memory_consolidation" }); + +function config(): OcxConfig { + return { + ...getDefaultConfig(), + defaultProvider: "gateway", + providers: { + gateway: { + adapter: "openai-responses", authMode: "key", + baseUrl: "https://gateway.example/v1", apiKey: "fixture-key", + }, + }, + memoryModels: { + extract: { model: "gateway/cheap", reasoningEffort: "high" }, + consolidation: { model: "gateway/strong", reasoningEffort: "xhigh" }, + }, + }; +} + +function body(model = "gpt-5.6-luna"): Record { + return { + model, stream: false, + reasoning: { effort: "low", summary: "auto" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Summarize this rollout." }] }], + }; +} + +function request(value: unknown, metadata?: string, extraHeaders: Record = {}): Request { + return new Request("http://localhost/v1/responses", { + method: "POST", + headers: { + "content-type": "application/json", session_id: "memory-models-fixture", + ...(metadata ? { "x-codex-turn-metadata": metadata } : {}), + ...extraHeaders, + }, + body: JSON.stringify(value), + }); +} + +function completion(): Record { + return { + id: "resp_memory_fixture", status: "completed", + output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "ok" }] }], + usage: { input_tokens: 10, output_tokens: 5, total_tokens: 15 }, + }; +} + +beforeEach(() => { + releaseSpendHome = acquireOwnedSpendHome(); +}); + +afterEach(() => { + releaseSpendHome?.(); + releaseSpendHome = undefined; + globalThis.fetch = originalFetch; + clearComboSelectionState(); + clearComboTargetCooldowns(); +}); + +describe("memory phase detection", () => { + test("recognizes each phase from Codex's own turn metadata", () => { + expect(detectMemoryModelPhase(body(), new Headers({ "x-codex-turn-metadata": extractMetadata() }))).toBe("extract"); + expect(detectMemoryModelPhase(body("gpt-5.6-terra"), new Headers({ "x-codex-turn-metadata": consolidationMetadata() }))).toBe("consolidation"); + }); + + test("takes the phase from the sub-agent header when the metadata copy carries none", () => { + const headers = new Headers({ + "x-codex-turn-metadata": JSON.stringify({ request_kind: "turn" }), + "x-openai-subagent": "memory_consolidation", + }); + expect(detectMemoryModelPhase(body("gpt-5.6-terra"), headers)).toBe("consolidation"); + // Any other internal turn category is not a memory turn. + expect(detectMemoryModelPhase(body(), new Headers({ "x-openai-subagent": "collab_spawn" }))).toBeNull(); + expect(detectMemoryModelPhase(body(), new Headers({ "x-openai-subagent": "review" }))).toBeNull(); + }); + + test("an ordinary turn, absent metadata, or malformed metadata is never a memory turn", () => { + expect(detectMemoryModelPhase(body(), new Headers())).toBeNull(); + expect(detectMemoryModelPhase(body(), new Headers({ "x-codex-turn-metadata": JSON.stringify({ request_kind: "turn", thread_source: "cli" }) }))).toBeNull(); + for (const value of ["{", "null", "[]", '"memory"', JSON.stringify({ request_kind: "memory_consolidation" })]) { + expect(detectMemoryModelPhase(body(), new Headers({ "x-codex-turn-metadata": value }))).toBeNull(); + } + // A non-string copy is malformed rather than absent. + expect(detectMemoryModelPhase({ ...body(), client_metadata: { "x-codex-turn-metadata": 42 } }, new Headers())).toBeNull(); + }); + + test("conflicting copies are not treated as a memory turn", () => { + for (const [header, embedded] of [[extractMetadata(), consolidationMetadata()], [consolidationMetadata(), extractMetadata()], [extractMetadata(), "{"], ["{", extractMetadata()]]) { + const input = { ...body(), client_metadata: { "x-codex-turn-metadata": embedded } }; + expect(detectMemoryModelPhase(input, new Headers({ "x-codex-turn-metadata": header! }))).toBeNull(); + } + }); + + test("both copies must agree on the same phase", () => { + const input = { ...body(), client_metadata: { "x-codex-turn-metadata": extractMetadata() } }; + expect(detectMemoryModelPhase(input, new Headers({ "x-codex-turn-metadata": extractMetadata() }))).toBe("extract"); + }); + + test("WebSocket frames read the body copy instead of the handshake header", () => { + const input = { ...body("gpt-5.6-terra"), client_metadata: { "x-codex-turn-metadata": consolidationMetadata() } }; + const headers = new Headers({ "x-codex-turn-metadata": extractMetadata() }); + expect(detectMemoryModelPhase(input, headers)).toBeNull(); + expect(detectMemoryModelPhase(input, headers, { transport: "websocket" })).toBe("consolidation"); + }); +}); + +describe("memory model settings", () => { + test("a phase without a model is off, and a blank model is not a destination", () => { + const settings = config(); + expect(configuredMemoryModel(settings, "extract")).toEqual({ model: "gateway/cheap", reasoningEffort: "high" }); + delete settings.memoryModels!.consolidation; + expect(configuredMemoryModel(settings, "consolidation")).toBeUndefined(); + settings.memoryModels = { extract: { model: " " } }; + expect(configuredMemoryModel(settings, "extract")).toBeUndefined(); + expect(configuredMemoryModel(undefined, "extract")).toBeUndefined(); + }); + + test("the configured effort is written to both wire shapes", () => { + const parsed = { modelId: "gpt-5.6-luna", options: { reasoning: "low" }, _rawBody: { reasoning: { effort: "low", summary: "auto" } } } as unknown as OcxParsedRequest; + expect(applyMemoryModelEffort(parsed, config(), "extract")).toEqual({ from: "low", to: "high" }); + expect(parsed.options.reasoning).toBe("high"); + expect(parsed._rawBody!.reasoning).toEqual({ effort: "high", summary: "auto" }); + // Idempotent, and a phase without an effort leaves Codex's own value alone. + expect(applyMemoryModelEffort(parsed, config(), "extract")).toBeNull(); + expect(applyMemoryModelEffort(parsed, config(), "consolidation")).toEqual({ from: "high", to: "xhigh" }); + const bare = config(); + bare.memoryModels = { extract: { model: "gateway/cheap" } }; + const untouched = { modelId: "gpt-5.6-luna", options: { reasoning: "low" }, _rawBody: {} } as unknown as OcxParsedRequest; + expect(applyMemoryModelEffort(untouched, bare, "extract")).toBeNull(); + expect(untouched.options.reasoning).toBe("low"); + }); +}); + +describe("memory model config", () => { + test("validates both phases without resetting providers on malformed hand edits", () => { + expect(validateConfigCandidate(config()).ok).toBe(true); + for (const value of [null, [], "cheap", { extract: { model: " " } }, { extract: { model: 42 } }, + { consolidation: { model: "gateway/strong", reasoningEffort: "fast" } }, + { extract: { model: "gateway/cheap", typo: true } }, { extract: {}, unknown: true }]) { + const raw = { ...config(), memoryModels: value }; + expect(validateConfigCandidate(raw).ok).toBe(false); + const loaded = configSchema.parse(raw); + expect(loaded.memoryModels).toBeUndefined(); + expect(loaded.providers).toEqual(config().providers); + } + }); + + test("an empty block is valid and means both phases stay with Codex", () => { + const raw = { ...config(), memoryModels: {} }; + expect(validateConfigCandidate(raw).ok).toBe(true); + expect(configuredMemoryModel(configSchema.parse(raw) as OcxConfig, "extract")).toBeUndefined(); + }); + + test("a dropped hand-edited block warns at load; valid or absent blocks stay silent", () => { + const warnings: string[] = []; + const original = console.warn; + console.warn = (message: unknown) => { warnings.push(String(message)); }; + try { + const invalid = { ...config(), memoryModels: { extract: { model: "" } } }; + warnDegradedMemoryModels(invalid, configSchema.parse(invalid) as OcxConfig); + expect(warnings).toHaveLength(1); + expect(warnings[0]).toContain("memoryModels is invalid"); + warnDegradedMemoryModels(config(), configSchema.parse(config()) as OcxConfig); + const absent = config(); + delete absent.memoryModels; + warnDegradedMemoryModels(absent, configSchema.parse(absent) as OcxConfig); + expect(warnings).toHaveLength(1); + } finally { + console.warn = original; + } + }); +}); + +describe("memory model routing", () => { + test("routes each phase to its own model and effort", async () => { + const settings = config(); + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + + const extractCtx = { model: "", provider: "" } as { model: string; provider: string; requestedModel?: string }; + const extract = await handleResponses(request(body(), extractMetadata()), settings, extractCtx); + expect(extract.status).toBe(200); + await extract.text(); + expect(calls[0]!.model).toBe("cheap"); + expect(calls[0]!.reasoning.effort).toBe("high"); + // The caller's own selector stays in the log; only the served model changed. + expect(extractCtx.requestedModel).toBe("gpt-5.6-luna"); + expect(extractCtx.model).toBe("cheap"); + + const consolidationCtx = { model: "", provider: "" } as { model: string; provider: string; requestedModel?: string }; + const consolidation = await handleResponses(request(body("gpt-5.6-terra"), consolidationMetadata()), settings, consolidationCtx); + expect(consolidation.status).toBe(200); + await consolidation.text(); + expect(calls[1]!.model).toBe("strong"); + expect(calls[1]!.reasoning.effort).toBe("xhigh"); + expect(consolidationCtx.requestedModel).toBe("gpt-5.6-terra"); + }); + + test("an unconfigured phase and a turn without the marker keep their own model", async () => { + const settings = config(); + delete settings.memoryModels!.consolidation; + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + + // Phase 2 with only Phase 1 configured, then a Phase 1 turn with nothing configured. + const consolidation = await handleResponses(request(body("gateway/normal"), consolidationMetadata()), settings, { model: "", provider: "" }); + expect(consolidation.status).toBe(200); + await consolidation.text(); + const configured = config(); + delete configured.memoryModels; + const unconfigured = await handleResponses(request(body("gateway/normal"), extractMetadata()), configured, { model: "", provider: "" }); + expect(unconfigured.status).toBe(200); + await unconfigured.text(); + // Same model id, no marker: nothing about the phase may reach it. + const ordinary = await handleResponses(request(body("gateway/normal")), settings, { model: "", provider: "" }); + expect(ordinary.status).toBe(200); + await ordinary.text(); + expect(calls.map(call => [call.model, call.reasoning.effort])).toEqual([ + ["normal", "low"], ["normal", "low"], ["normal", "low"], + ]); + }); + + test("a memory turn keeps the phase decision when the shadow intercept would match too", async () => { + const settings = config(); + settings.shadowCallIntercept = { enabled: true, model: "gateway/helper" }; + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + const logCtx = { model: "", provider: "" } as { model: string; provider: string; shadowCallRewrittenFrom?: string }; + const response = await handleResponses(request(body(), extractMetadata()), settings, logCtx); + expect(response.status).toBe(200); + await response.text(); + expect(calls[0]!.model).toBe("cheap"); + expect(calls[0]!.reasoning.effort).toBe("high"); + // Phase 1 shares its model id with the app's helper calls, so the marker is what tells them apart. + expect(logCtx.shadowCallRewrittenFrom).toBeUndefined(); + }); + + test("a target that no longer resolves fails the memory call instead of falling back", async () => { + const settings = config(); + settings.memoryModels = { extract: { model: "ghost/cheap" } }; + const calls: string[] = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body)).model); + return Response.json(completion()); + }) as typeof fetch; + const response = await handleResponses(request(body(), extractMetadata()), settings, { model: "", provider: "" }); + expect(response.status).toBe(409); + expect((await response.json() as { error: { code: string } }).error.code).toBe(MEMORY_MODEL_TARGET_UNAVAILABLE_CODE); + expect(calls).toEqual([]); + }); + + test("the phase decision survives the combo handoff", async () => { + const settings = config(); + settings.combos = { memory: { targets: [{ provider: "gateway", model: "cheap" }] } }; + settings.memoryModels = { extract: { model: "combo/memory", reasoningEffort: "high" } }; + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + const logCtx = { model: "", provider: "" } as { model: string; provider: string; requestedModel?: string }; + const response = await handleResponses(request(body(), extractMetadata()), settings, logCtx); + expect(response.status).toBe(200); + await response.text(); + expect(calls[0]!.model).toBe("cheap"); + expect(calls[0]!.reasoning.effort).toBe("high"); + // A combo target has to reach the dispatcher as `model`, so the rewritten selector is what the + // log records as requested; the phase itself is named in the route decision. + expect(logCtx.requestedModel).toBe("combo/memory"); + }); +}); From 6ffadfdc0c28b3d1aaadce7d686715cf51866277 Mon Sep 17 00:00:00 2001 From: Robin Bially <7304732+RobinBially@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:13:22 +0200 Subject: [PATCH 02/13] fix(gui): give the memory pickers one fixed width and center them in their rows The two pickers were sized by their own labels, so a long model id next to "Medium" produced two differently sized pills, and both sat at the top of the row instead of centred on the copy. The pair is now one band with two equal shares: a model id has to stay readable (14rem holds "opencode-go/glm-5.3-flash" with room to spare, the ceiling the sidecar pickers already use), so that width drives both, and the row centres the band on the copy. Also: the info glyph loses the focus ring the shared rule drew around a single character and highlights itself on hover and keyboard focus instead, the account notice no longer claims there is no choice while a phase is routed, and the new memory-models module joins the responses-core owner roster the full suite checks. --- gui/src/components/MemoryModelsPanel.tsx | 18 +++++--- gui/src/i18n/de.ts | 2 +- gui/src/i18n/en.ts | 2 +- gui/src/i18n/fr.ts | 2 +- gui/src/i18n/ja.ts | 2 +- gui/src/i18n/ko.ts | 2 +- gui/src/i18n/ru.ts | 2 +- gui/src/i18n/tr.ts | 2 +- gui/src/i18n/vi.ts | 2 +- gui/src/i18n/zh-TW.ts | 2 +- gui/src/i18n/zh.ts | 2 +- gui/src/styles-dashboard-workspace.css | 59 ++++++++++++++++++++++++ tests/helpers/responses-core-source.ts | 1 + 13 files changed, 81 insertions(+), 17 deletions(-) diff --git a/gui/src/components/MemoryModelsPanel.tsx b/gui/src/components/MemoryModelsPanel.tsx index dcd00d5e478..33d5f7f9e27 100644 --- a/gui/src/components/MemoryModelsPanel.tsx +++ b/gui/src/components/MemoryModelsPanel.tsx @@ -137,15 +137,18 @@ function MemoryModelsControls({ apiBase, models }: { apiBase: string; models: Mo || extractEffort !== (saved?.extract?.reasoningEffort ?? "") || consolidationModel !== (saved?.consolidation?.model ?? "") || consolidationEffort !== (saved?.consolidation?.reasoningEffort ?? ""); + // The account notice is about the phase that stays on Codex's own model, so it is both + // true and useful only while exactly one of the two phases is routed. + const partiallyRouted = Boolean(extractModel) !== Boolean(consolidationModel); const info = t("memoryModels.info"); const row = (phase: Phase, model: string, effort: string, setModel: (value: string) => void, setEffort: (value: string) => void) => ( -
+
{t(`memoryModels.${phase}` as TKey)}
{t(`memoryModels.${phase}Hint` as TKey)}
-
+