diff --git a/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md b/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md index 3fb8774a8ce..ae13d197cce 100644 --- a/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/fr/reference/cli/providers-accounts.md @@ -415,7 +415,7 @@ proxy en cours d'exécution (`ocx start` ou un service installé). | `provider ` | `--json` | Activez ou désactivez chaque modèle d'un fournisseur en une seule écriture. | | `selected ` | `--set `, `--clear`, `--json` | Lisez ou remplacez la liste autorisée du modèle de fournisseur. `--clear` supprime la liste blanche afin que chaque modèle soit proposé. | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` | Lisez ou définissez la limite de la fenêtre contextuelle, globalement ou par fournisseur. `value --set-all` redirige également chaque fournisseur acheminé (comme la bascule du tableau de bord) ; sans cela, la valeur devient uniquement la valeur par défaut. `provider ... on --value ` définit un plafond explicite pour ce fournisseur uniquement (`--value` est valide avec `on` uniquement). | -| `shadow [model\|-]` | `--enabled `, `--json` | Lisez ou définissez le modèle de remplacement pour les appels d'assistance en arrière-plan de Codex. `-` efface le modèle. `status` signale également `sourceModels`, l'assistant supprime les interceptions du proxy (par défaut : `gpt-6-luna`, `gpt-5.6-luna` ; les clients via 0.144.x ont utilisé `gpt-5.4-mini`, qu'une substitution explicite de `sourceModels` peut restaurer). | +| `shadow [model\|-]` | `--enabled `, `--json` | Lisez ou définissez le modèle de remplacement pour les appels d'assistance en arrière-plan de Codex. `-` efface le modèle. `status` signale également `sourceModels`, l'assistant supprime les interceptions du proxy (par défaut : `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra` ; les clients via 0.144.x ont utilisé `gpt-5.4-mini`, qu'une substitution explicite de `sourceModels` peut restaurer). | ```bash ocx models live --json # what Codex can actually see right now diff --git a/docs-site/src/content/docs/fr/reference/configuration/server.md b/docs-site/src/content/docs/fr/reference/configuration/server.md index 3a9ca385079..a9ae3c64222 100644 --- a/docs-site/src/content/docs/fr/reference/configuration/server.md +++ b/docs-site/src/content/docs/fr/reference/configuration/server.md @@ -26,7 +26,7 @@ exécute des fonctionnalités d'assistance autour des demandes du fournisseur. | `codexAutoStart?` | `boolean` | `true` | Autorise le lanceur intermédiaire Codex à exécuter `ocx ensure` avant de démarrer Codex. Avec la valeur false, cette vérification ne fait rien. | | `codexShimAutoRestore?` | `boolean` | `true` | Restaure le lanceur intermédiaire installé après son remplacement par une mise à jour externe de Codex terminée. Désactivation par variable d'environnement : `OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`. | | `syncResumeHistory?` | `boolean` | `true` | Compatibilité historique Codex App réversible. Les métadonnées originales sont sauvegardées et restaurées par `ocx stop` / `ocx restore`. | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | désactivé | Redirigez les appels Codex helper/shadow reconnus vers un modèle choisi tout en conservant l'effort de raisonnement configuré pour la requête. Le préfixe source par défaut est `gpt-6-luna`, `gpt-5.6-luna` ; les clients plus anciens via 0.144.x utilisaient `gpt-5.4-mini`, que `sourceModels` peut restaurer. | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | désactivé | Redirigez les appels Codex helper/shadow reconnus vers un modèle choisi tout en conservant l'effort de raisonnement configuré pour la requête. Le préfixe source par défaut est `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra` ; les clients plus anciens via 0.144.x utilisaient `gpt-5.4-mini`, que `sourceModels` peut restaurer. | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | activé lorsqu'il est utilisable | Options du service auxiliaire de recherche Web. | | `visionSidecar?` | `OcxVisionSidecarConfig` | activé lorsqu'il est utilisable | Options du service auxiliaire de description d'images. | | `images?` | `OcxImagesConfig` | sélection automatique OpenAI | Options de relais d'images autonomes pour Codex `image_gen`. | @@ -194,7 +194,7 @@ l'en-tête JSON `x-codex-turn-metadata` sont exemptées, afin qu'un sous-agent e "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md index 479fac0f0da..1767f3b9148 100644 --- a/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ja/reference/cli/providers-accounts.md @@ -305,7 +305,7 @@ native-main トラフィックまたはジャーナル復旧を受け入れる | `provider ` | `--json` | 1 つのプロバイダーのすべてのモデルを 1 回の書き込みで有効または無効にします。 | | `selected ` | `--set `、`--clear`、`--json` |プロバイダー モデルのホワイトリストを読み取るか置き換えます。 `--clear` はホワイトリストを削除し、すべてのモデルが提供されるようにします。 | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` |コンテキスト ウィンドウ キャップをグローバルに、またはプロバイダーごとに読み取りまたは設定します。 `value --set-all` はすべてのルーティング済みプロバイダーにも値を再適用します(ダッシュボードのトグルと同様)。指定しない場合は既定値のみが変更されます。 `provider ... on --value ` はそのプロバイダーのみに個別のキャップを設定します(`--value` は `on` でのみ使用できます)。 | -| `shadow [model\|-]` | `--enabled `、`--json` | Codex のバックグラウンド ヘルパー呼び出しの置換モデルを読み取るか、設定します。 `-` はモデルをクリアします。 `status` は `sourceModels` も報告し、プロキシがインターセプトするヘルパースラッグを示します (デフォルト: `gpt-6-luna`, `gpt-5.6-luna`; 0.144.x 以前のクライアントが使用した `gpt-5.4-mini` は明示的な `sourceModels` オーバーライドで復元できます)。 | +| `shadow [model\|-]` | `--enabled `、`--json` | Codex のバックグラウンド ヘルパー呼び出しの置換モデルを読み取るか、設定します。 `-` はモデルをクリアします。 `status` は `sourceModels` も報告し、プロキシがインターセプトするヘルパースラッグを示します (デフォルト: `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`; 0.144.x 以前のクライアントが使用した `gpt-5.4-mini` は明示的な `sourceModels` オーバーライドで復元できます)。 | ```bash ocx models live --json # what Codex can actually see right now diff --git a/docs-site/src/content/docs/ja/reference/configuration/server.md b/docs-site/src/content/docs/ja/reference/configuration/server.md index 27dbf08b986..81191aea260 100644 --- a/docs-site/src/content/docs/ja/reference/configuration/server.md +++ b/docs-site/src/content/docs/ja/reference/configuration/server.md @@ -26,7 +26,7 @@ description: リスナー、リモート アクセス、アドミッション | `codexAutoStart?` | `boolean` | `true` | Codex を起動する前に、Codex シムで `ocx ensure` を実行させます。 False を指定すると、操作が行われないことが保証されます。 | | `codexShimAutoRestore?` | `boolean` | `true` |完了した外部 Codex アップデートによってインストールされたシムが置き換えられた後、インストールされているシムを復元します。環境オプトアウト: `OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`。 | | `syncResumeHistory?` | `boolean` | `true` | Codex App 履歴の互換性を元に戻すことができます。元のメタデータは `ocx stop` / `ocx restore` によってバックアップおよび復元されます。 | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` |オフ |認識された Codex ヘルパー/シャドウ呼び出しを、リクエストに設定された推論エフォートを維持したまま選択したモデルにリダイレクトします。デフォルトのソースプレフィックスは `gpt-6-luna`, `gpt-5.6-luna` です。0.144.x 以前のクライアントでは `gpt-5.4-mini` が使われており、`sourceModels` で復元できます。 | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` |オフ |認識された Codex ヘルパー/シャドウ呼び出しを、リクエストに設定された推論エフォートを維持したまま選択したモデルにリダイレクトします。デフォルトのソースプレフィックスは `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra` です。0.144.x 以前のクライアントでは `gpt-5.4-mini` が使われており、`sourceModels` で復元できます。 | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` |使用可能な場合はオン | Web 検索サイドカー オプション。 | | `visionSidecar?` | `OcxVisionSidecarConfig` |使用可能な場合はオン |画像説明サイドカー オプション。 | | `images?` | `OcxImagesConfig` | OpenAI の自動選択 | Codex `image_gen` のスタンドアロン イメージ リレー オプション。 | @@ -121,7 +121,7 @@ Codex は、タイトルやコミット メッセージなどのタスクに小 "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md index 3a395cca85d..86244be209f 100644 --- a/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ko/reference/cli/providers-accounts.md @@ -389,7 +389,7 @@ native-main 트래픽이나 저널 복구를 허용하기 전에 수명 주기 | `provider ` | `--json` | 한 제공자의 모든 모델을 한 번의 쓰기로 활성화하거나 비활성화합니다. | | `selected ` | `--set `, `--clear`, `--json` | 제공자 모델 허용 목록을 읽거나 교체합니다. `--clear`는 허용 목록을 제거해 모든 모델을 제공하도록 합니다. | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` | 전역 또는 제공자별로 컨텍스트 창 한도를 읽거나 설정합니다. `value --set-all`은 모든 라우팅된 공급자에도 값을 다시 적용합니다(대시보드 토글과 동일). 지정하지 않으면 값은 기본값만 변경됩니다. `provider ... on --value `는 해당 제공자에만 별도 한도를 설정합니다(`--value`는 `on`에서만 사용할 수 있습니다). | -| `shadow [model\|-]` | `--enabled `, `--json` | Codex의 백그라운드 헬퍼 호출에 사용할 대체 모델을 읽거나 설정합니다. `-`는 모델을 지웁니다. `status`는 프록시가 가로채는 헬퍼 슬러그인 `sourceModels`도 보고합니다(기본값: `gpt-6-luna`, `gpt-5.6-luna`; 0.144.x 이하 클라이언트가 사용한 `gpt-5.4-mini`는 명시적인 `sourceModels` 재정의로 복원할 수 있습니다). | +| `shadow [model\|-]` | `--enabled `, `--json` | Codex의 백그라운드 헬퍼 호출에 사용할 대체 모델을 읽거나 설정합니다. `-`는 모델을 지웁니다. `status`는 프록시가 가로채는 헬퍼 슬러그인 `sourceModels`도 보고합니다(기본값: `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`; 0.144.x 이하 클라이언트가 사용한 `gpt-5.4-mini`는 명시적인 `sourceModels` 재정의로 복원할 수 있습니다). | ```bash ocx models live --json # what Codex can actually see right now diff --git a/docs-site/src/content/docs/ko/reference/configuration/server.md b/docs-site/src/content/docs/ko/reference/configuration/server.md index b2a71843a4e..fdd73ba194f 100644 --- a/docs-site/src/content/docs/ko/reference/configuration/server.md +++ b/docs-site/src/content/docs/ko/reference/configuration/server.md @@ -26,7 +26,7 @@ description: 리스너, 원격 접근, admission 키, 타임아웃, 저장소, | `codexAutoStart?` | `boolean` | `true` | Codex shim이 Codex를 실행하기 전에 `ocx ensure`를 돌리도록 허용합니다. `false`이면 ensure는 아무 작업도 하지 않습니다. | | `codexShimAutoRestore?` | `boolean` | `true` | 완료된 외부 Codex 업데이트가 설치된 shim을 교체한 뒤 복원합니다. 환경 변수로 끌 수 있습니다: `OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`. | | `syncResumeHistory?` | `boolean` | `true` | 되돌릴 수 있는 Codex App history 호환성입니다. 원래 메타데이터는 `ocx stop` / `ocx restore`가 백업하고 복원합니다. | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | 인식된 Codex 보조/섀도 호출을 요청에 설정된 reasoning effort를 유지한 채 선택한 모델로 다시 보냅니다. 기본 source prefix는 `gpt-6-luna`, `gpt-5.6-luna`입니다. 0.144.x 이하의 이전 클라이언트는 `gpt-5.4-mini`를 사용했으며 `sourceModels`로 복원할 수 있습니다. | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | 인식된 Codex 보조/섀도 호출을 요청에 설정된 reasoning effort를 유지한 채 선택한 모델로 다시 보냅니다. 기본 source prefix는 `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`입니다. 0.144.x 이하의 이전 클라이언트는 `gpt-5.4-mini`를 사용했으며 `sourceModels`로 복원할 수 있습니다. | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | on when usable | 웹 검색 사이드카 옵션입니다. | | `visionSidecar?` | `OcxVisionSidecarConfig` | on when usable | 이미지 설명 사이드카 옵션입니다. | | `images?` | `OcxImagesConfig` | automatic OpenAI selection | Codex `image_gen`용 독립형 Images 릴레이 옵션입니다. | @@ -169,7 +169,7 @@ Codex는 제목과 커밋 메시지 같은 작업에 작은 보조 모델을 사 "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/reference/cli/providers-accounts.md b/docs-site/src/content/docs/reference/cli/providers-accounts.md index 93abe731d2e..8db415a9910 100644 --- a/docs-site/src/content/docs/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/reference/cli/providers-accounts.md @@ -690,7 +690,7 @@ proxy to be running (`ocx start`, or an installed service). | `provider ` | `--json` | Enable or disable every model of one provider in a single write. | | `selected ` | `--set `, `--clear`, `--json` | Read or replace the provider model allowlist. `--clear` removes the allowlist so every model is offered. | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` | Read or set the context-window cap, globally or per provider. `value --set-all` also re-points every routed provider (like the dashboard toggle); without it the value only becomes the default. `provider ... on --value ` sets an explicit cap for that provider only (`--value` is valid with `on` only). | -| `shadow [model\|-]` | `--enabled `, `--json` | Read or set the replacement model for Codex's background helper calls. `-` clears the model. `status` also reports `sourceModels`, the helper slugs the proxy intercepts (default: `gpt-6-luna`, `gpt-5.6-luna`; clients through 0.144.x used `gpt-5.4-mini`, which an explicit `sourceModels` override can restore). | +| `shadow [model\|-]` | `--enabled `, `--json` | Read or set the replacement model for Codex's background helper calls. `-` clears the model. `status` also reports `sourceModels`, the helper slugs the proxy intercepts (default: `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`; clients through 0.144.x used `gpt-5.4-mini`, which an explicit `sourceModels` override can restore). | ```bash ocx models live --json # what Codex can actually see right now diff --git a/docs-site/src/content/docs/reference/configuration/server.md b/docs-site/src/content/docs/reference/configuration/server.md index c8e9f967ffb..2a13064ab0f 100644 --- a/docs-site/src/content/docs/reference/configuration/server.md +++ b/docs-site/src/content/docs/reference/configuration/server.md @@ -40,7 +40,8 @@ runs helper features around provider requests. | `codexProviderDisplayName?` | `string` | `"OpenCodex Proxy"` | Label Codex shows for the injected `opencodex` provider, written as its `name` field in `config.toml` and the reference profile. Presentation only: routing resolves through the provider id `opencodex`, so a rename never moves `model_provider = "opencodex"` or the `[model_providers.opencodex]` header and cannot orphan threads already tagged with that id. Codex refuses to load a provider with no name, so there is no way to omit the field — choose a neutral label instead. A blank, over-128-character, or control-character value is ignored and the default label is written. | | `resetCreditAutoRedeem?` | `{ enabled?: boolean; leadTimeMinutes?: number }` | off | Opt-in: redeem the main Codex account's soonest-expiring reset credit `leadTimeMinutes` (1–60, default 10) before it expires. Every attempt re-reads the upstream credit list first and skips when the credit is gone (for example, redeemed by hand); the `redeem_request_id` is journaled in `$OPENCODEX_HOME/reset-credit-auto-redeem.json` before the call so a crash replays the same idempotent request instead of spending a second credit. Servers sharing this configuration directory coordinate reservations and settlements so one process does not replace another's request record. Logs carry a hashed account key only. | | `syncResumeHistory?` | `boolean` | `true` | Reversible Codex App history compatibility. Original metadata is backed up and restored by `ocx stop` / `ocx restore`. | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | Redirect recognized Codex helper/shadow calls to a chosen model while preserving the request's configured reasoning effort. The default source prefixes are `gpt-6-luna` and `gpt-5.6-luna`; older clients through 0.144.x used `gpt-5.4-mini`, which `sourceModels` can restore. | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | Redirect recognized Codex helper/shadow calls to a chosen model while preserving the request's configured reasoning effort. The default source prefixes are `gpt-6-luna`, `gpt-5.6-luna`, and `gpt-5.6-terra`, the model Codex asks for its background memory-consolidation pass; older clients through 0.144.x used `gpt-5.4-mini`, which `sourceModels` can restore. | +| `memoryModels?` | `{ extract?: { model: string; reasoningEffort?: string }; consolidation?: { model: string; reasoningEffort?: string } }` | off | Route Codex's two memory phases to a chosen model, with an optional reasoning effort per phase. See [Memory routing](#memory-routing). | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | on when usable | Web-search sidecar options. | | `visionSidecar?` | `OcxVisionSidecarConfig` | on when usable | Image-description sidecar options. | | `images?` | `OcxImagesConfig` | automatic OpenAI selection | Standalone Images relay options for Codex `image_gen`. | @@ -658,14 +659,66 @@ caller's credential does not cross to the other provider. The selected model mus input size and content. Restart the proxy after editing `config.json` by hand. Dashboard saves apply immediately. +## Memory routing + +In **Dashboard → Overview → Memory routing**, choose a model and an optional reasoning effort for +each of Codex's two memory phases, then click **Save**. Select **Off — Codex default** and save to +remove the override. Changes apply to the next memory request without restarting the proxy. + +Set `memoryModels` in OpenCodex `config.json` to route those requests. Both phases keep Codex's own +model while the block is omitted, and the phases are independent: configuring one leaves the other +alone. + +```json +{ + "memoryModels": { + "extract": { "model": "provider/model-id", "reasoningEffort": "low" }, + "consolidation": { "model": "provider/model-id", "reasoningEffort": "medium" } + } +} +``` + +`extract` is the pass that summarizes one finished session into a raw memory; `consolidation` is the +single agent run that merges those raw memories into the files under `$CODEX_HOME/memories`. +`model` accepts native model IDs, provider-qualified model IDs, and configured combos. +`reasoningEffort` is optional; omit it to keep the effort Codex asked for. Supported declarations are +`none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`, and `ultra`. Codex hard-codes `low` for +extract and `medium` for consolidation, so a configured effort replaces that value. + +OpenCodex recognizes these requests from Codex's own turn metadata: `request_kind: "memory"` in +the `x-codex-turn-metadata` header marks an extract pass, and `thread_source: +"memory_consolidation"` marks the consolidation thread. On HTTP, a request whose +`x-openai-subagent` header names `memory_consolidation` counts as a consolidation pass on its +own. The model id is deliberately not a signal: the extract pass runs on the same helper model +Codex uses for titles and commit messages, so a model-based rule would also capture ordinary +helper calls. Missing, malformed, or conflicting metadata does not activate the override; when +several copies of the metadata are supplied they must name the same phase. WebSocket requests use +each frame's metadata rather than the connection's earlier handshake metadata — the bridge +re-attaches the handshake's `x-openai-subagent` header to every frame, so that header names the +connection, not the current pass, and is not a websocket signal. + +A configured phase wins when `shadowCallIntercept` would match the same request. A phase left off +keeps its current routing, which includes the shadow-call intercept: both phases run on default +intercept source models (`gpt-5.6-luna` for extract, `gpt-5.6-terra` for consolidation), so an +enabled intercept already covers them. The selected model's provider receives the session +text Codex summarizes for memory, including sessions that normally run on another provider; the +dashboard panel states this next to the model pickers. Without a choice, memory requests reach your +OpenAI account like any other native model. A phase whose target stopped resolving — the provider is +disabled or deleted, or its combo no longer exists — fails that memory call with `409` and error code +`memory_model_target_unavailable` instead of falling back to the default provider. The request log +names the phase (`memory-extract` or `memory-consolidation`) as the routing reason. Restart the +proxy after editing `config.json` by hand. Dashboard saves apply immediately. ## Shadow calls Codex uses small helper models for tasks such as titles and commit messages. Enable `shadowCallIntercept` to redirect recognized source-model prefixes to another configured model. The -replacement keeps the request's configured reasoning effort. Set `sourceModels` only when a client -uses different helper ids. A non-empty `sourceModels` replaces the default prefixes instead of -extending them, so include `gpt-6-luna` (and `gpt-5.6-luna` for 0.145.0-0.153.x clients) in the -list when current clients should still be intercepted. +replacement keeps the request's configured reasoning effort. The defaults also cover +`gpt-5.6-terra`, the model Codex asks for its background memory-consolidation pass, so an install +that intercepts helper traffic keeps the whole memory pipeline off the native account instead of +leaving that one phase on the route the operator moved away from. Set `sourceModels` only when a +client uses different helper ids. A non-empty `sourceModels` replaces the default prefixes instead +of extending them, so include `gpt-6-luna` and `gpt-5.6-terra` (plus `gpt-5.6-luna` for +0.145.0-0.153.x clients) in the list when current clients should still be intercepted. Interception is model-based: every request whose bare model id matches `sourceModels` can be redirected, including normal `request_kind: "turn"` requests. Requests marked as spawned children by `x-openai-subagent: collab_spawn` or `subagent_kind: "thread_spawn"` in the `x-codex-turn-metadata` @@ -676,7 +729,7 @@ JSON header are exempt, so an explicitly spawned sub-agent keeps its model. "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md b/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md index 7339d3f53bb..0abae4c4b08 100644 --- a/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/ru/reference/cli/providers-accounts.md @@ -372,7 +372,7 @@ management API и требуют, чтобы прокси уже работал | `provider ` | `--json` | Включить или выключить сразу все модели одного провайдера одним действием. | | `selected ` | `--set `, `--clear`, `--json` | Прочитать или заменить allowlist моделей провайдера. `--clear` удаляет allowlist, и тогда доступны все модели. | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` | Прочитать или задать context-window cap глобально либо по провайдерам. `value --set-all` также переустанавливает значение для всех маршрутизируемых провайдеров (как переключатель дашборда); без него меняется только значение по умолчанию. `provider ... on --value ` задаёт отдельный cap только для этого провайдера (`--value` допустим только с `on`). | -| `shadow [model\|-]` | `--enabled `, `--json` | Прочитать или задать модель-замену для background helper-call'ов Codex. `-` очищает модель. `status` также показывает `sourceModels` — helper-slug'и, которые перехватывает proxy (по умолчанию `gpt-6-luna`, `gpt-5.6-luna`; `gpt-5.4-mini` для клиентов до 0.144.x включительно можно восстановить явным переопределением `sourceModels`). | +| `shadow [model\|-]` | `--enabled `, `--json` | Прочитать или задать модель-замену для background helper-call'ов Codex. `-` очищает модель. `status` также показывает `sourceModels` — helper-slug'и, которые перехватывает proxy (по умолчанию `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`; `gpt-5.4-mini` для клиентов до 0.144.x включительно можно восстановить явным переопределением `sourceModels`). | ```bash ocx models live --json # what Codex can actually see right now diff --git a/docs-site/src/content/docs/ru/reference/configuration/server.md b/docs-site/src/content/docs/ru/reference/configuration/server.md index f2a7bc704ab..9ad56573421 100644 --- a/docs-site/src/content/docs/ru/reference/configuration/server.md +++ b/docs-site/src/content/docs/ru/reference/configuration/server.md @@ -27,7 +27,7 @@ description: Listener, удалённый доступ, admission key, тайм | `codexAutoStart?` | `boolean` | `true` | Разрешает shim'у Codex запускать `ocx ensure` перед стартом Codex. При false `ensure` становится no-op. | | `codexShimAutoRestore?` | `boolean` | `true` | Восстанавливает установленный shim после завершённого внешнего обновления Codex, которое заменило его. Для отключения через окружение: `OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`. | | `syncResumeHistory?` | `boolean` | `true` | Обратимый режим совместимости истории Codex App. Исходные metadata резервируются и восстанавливаются через `ocx stop` / `ocx restore`. | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | Перенаправляет распознанные helper/shadow-call'ы Codex на выбранную модель с сохранением настроенного для запроса reasoning effort. Source-prefix по умолчанию: `gpt-6-luna`, `gpt-5.6-luna`; клиенты до 0.144.x включительно использовали `gpt-5.4-mini`, который можно восстановить через `sourceModels`. | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | Перенаправляет распознанные helper/shadow-call'ы Codex на выбранную модель с сохранением настроенного для запроса reasoning effort. Source-prefix по умолчанию: `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`; клиенты до 0.144.x включительно использовали `gpt-5.4-mini`, который можно восстановить через `sourceModels`. | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | on when usable | Настройки sidecar'а web-search. | | `visionSidecar?` | `OcxVisionSidecarConfig` | on when usable | Настройки sidecar'а описания изображений. | | `images?` | `OcxImagesConfig` | automatic OpenAI selection | Настройки standalone Images relay для Codex `image_gen`. | @@ -148,7 +148,7 @@ Codex использует маленькие helper-model'и для задач "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md b/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md index 1e5e797f963..64c8890b877 100644 --- a/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/tr/reference/cli/providers-accounts.md @@ -480,7 +480,7 @@ kurulu bir servis). | `provider ` | `--json` | Tek bir yazmada bir sağlayıcının her modelini etkinleştirin veya devre dışı bırakın. | | `selected ` | `--set `, `--clear`, `--json` | Sağlayıcı model izin listesini okuyun veya değiştirin. `--clear` her modelin sunulması için izin listesini kaldırır. | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` | Küresel olarak veya sağlayıcı başına bağlam penceresi sınırını okuyun veya ayarlayın. `value --set-all` ayrıca her yönlendirilen sağlayıcıyı yeniden yönlendirir (kontrol paneli anahtarı gibi); bu olmadan değer yalnızca varsayılan olur. `provider ... on --value ` yalnızca o sağlayıcı için açık bir sınır belirler (`--value` yalnızca `on` ile geçerlidir). | -| `shadow [model\|-]` | `--enabled `, `--json` | Codex'in arka plan yardımcı çağrıları için değiştirme modelini okuyun veya ayarlayın. `-` modeli temizler. `status` ayrıca proxy'nin müdahale ettiği yardımcı slug'ları olan `sourceModels`'ı bildirir (varsayılan: `gpt-6-luna`, `gpt-5.6-luna`; 0.144.x'e kadar olan istemciler açık bir `sourceModels` geçersiz kılmasının geri yükleyebileceği `gpt-5.4-mini` kullanmıştır). | +| `shadow [model\|-]` | `--enabled `, `--json` | Codex'in arka plan yardımcı çağrıları için değiştirme modelini okuyun veya ayarlayın. `-` modeli temizler. `status` ayrıca proxy'nin müdahale ettiği yardımcı slug'ları olan `sourceModels`'ı bildirir (varsayılan: `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`; 0.144.x'e kadar olan istemciler açık bir `sourceModels` geçersiz kılmasının geri yükleyebileceği `gpt-5.4-mini` kullanmıştır). | ```bash ocx models live --json # Codex'in şu anda gerçekte görebildikleri diff --git a/docs-site/src/content/docs/tr/reference/configuration/server.md b/docs-site/src/content/docs/tr/reference/configuration/server.md index 71524df464a..567c1b68835 100644 --- a/docs-site/src/content/docs/tr/reference/configuration/server.md +++ b/docs-site/src/content/docs/tr/reference/configuration/server.md @@ -27,7 +27,7 @@ yardımcı özellikleri nasıl çalıştıracağını kontrol eder. | `codexAutoStart?` | `boolean` | `true` | Codex dolgusunun Codex'i başlatmadan önce `ocx ensure` çalıştırmasına izin verin. False, ensure'ı bir işlem yapmayan (no-op) hale getirir. | | `codexShimAutoRestore?` | `boolean` | `true` | Tamamlanan harici bir Codex güncellemesi değiştirdikten sonra kurulu bir dolguyu geri yükleyin. Ortam vazgeçmesi: `OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`. | | `syncResumeHistory?` | `boolean` | `true` | Tersine çevrilebilir Codex App geçmişi uyumluluğu. Orijinal meta veriler yedeklenir ve `ocx stop` / `ocx restore` tarafından geri yüklenir. | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | kapalı | Tanınan Codex yardımcı/gölge çağrılarını, istek için yapılandırılan akıl yürütme çabasını koruyarak seçilen bir modele yeniden yönlendirin. Varsayılan kaynak öneki `gpt-6-luna`, `gpt-5.6-luna`'dır; 0.144.x'e kadar olan eski istemciler `sourceModels`'ın geri yükleyebileceği `gpt-5.4-mini` kullanmıştır. | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | kapalı | Tanınan Codex yardımcı/gölge çağrılarını, istek için yapılandırılan akıl yürütme çabasını koruyarak seçilen bir modele yeniden yönlendirin. Varsayılan kaynak öneki `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`'dır; 0.144.x'e kadar olan eski istemciler `sourceModels`'ın geri yükleyebileceği `gpt-5.4-mini` kullanmıştır. | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | kullanılabilir olduğunda açık | Web arama sidecar seçenekleri. | | `visionSidecar?` | `OcxVisionSidecarConfig` | kullanılabilir olduğunda açık | Görsel açıklama sidecar seçenekleri. | | `images?` | `OcxImagesConfig` | otomatik OpenAI seçimi | Codex `image_gen` için bağımsız Görseller aktarma seçenekleri. | @@ -217,7 +217,7 @@ bir alt aracının modeli korunur. "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md b/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md index 2a102ecec5c..6b44613e6b8 100644 --- a/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/zh-cn/reference/cli/providers-accounts.md @@ -340,7 +340,7 @@ v1 恢复矩阵覆盖的是事务文件通过重命名发布后 OpenCodex 进程 | `provider ` | `--json` | 一次写入中启用或禁用某个提供方的全部模型。 | | `selected ` | `--set `, `--clear`, `--json` | 读取或替换提供方模型允许列表。`--clear` 会移除允许列表,使所有模型都可提供。 | | `context [--set-all]\|provider on [--value ]\|provider off\|all >` | `--json` | 读取或设置上下文窗口上限,可全局设置或按提供方设置。`value --set-all` 还会把值重新应用到所有已路由提供方(等同于仪表板开关);不加它则只改变默认值。`provider ... on --value ` 仅为该提供方设置独立上限(`--value` 仅可用于 `on`)。 | -| `shadow [model\|-]` | `--enabled `, `--json` | 读取或设置 Codex 后台辅助调用所替换的模型。`-` 会清除该模型。`status` 还会报告 `sourceModels`,即代理拦截的辅助器 slug(默认值:`gpt-6-luna`, `gpt-5.6-luna`;0.144.x 及更早客户端使用的 `gpt-5.4-mini` 可通过显式 `sourceModels` 覆盖恢复)。 | +| `shadow [model\|-]` | `--enabled `, `--json` | 读取或设置 Codex 后台辅助调用所替换的模型。`-` 会清除该模型。`status` 还会报告 `sourceModels`,即代理拦截的辅助器 slug(默认值:`gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`;0.144.x 及更早客户端使用的 `gpt-5.4-mini` 可通过显式 `sourceModels` 覆盖恢复)。 | ```bash ocx models live --json # what Codex can actually see right now diff --git a/docs-site/src/content/docs/zh-cn/reference/configuration/server.md b/docs-site/src/content/docs/zh-cn/reference/configuration/server.md index 76efbe5738e..aee2372e666 100644 --- a/docs-site/src/content/docs/zh-cn/reference/configuration/server.md +++ b/docs-site/src/content/docs/zh-cn/reference/configuration/server.md @@ -27,7 +27,7 @@ description: 监听、远程访问、准入密钥、超时、存储、侧车、 | `codexAutoStart?` | `boolean` | `true` | 允许 Codex shim 在启动 Codex 之前运行 `ocx ensure`。设为 false 会让 ensure 变成无操作。 | | `codexShimAutoRestore?` | `boolean` | `true` | 在完成外部 Codex 更新并覆盖安装的 shim 之后恢复该 shim。环境退出开关:`OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`。 | | `syncResumeHistory?` | `boolean` | `true` | 可逆的 Codex App 历史兼容性。原始元数据会被备份,并由 `ocx stop` / `ocx restore` 恢复。 | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | 将识别出的 Codex 辅助/影子调用重定向到选定模型,并保留为请求配置的推理强度。默认源前缀为 `gpt-6-luna`, `gpt-5.6-luna`;0.144.x 及更早客户端使用 `gpt-5.4-mini`,可通过 `sourceModels` 恢复。 | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | 将识别出的 Codex 辅助/影子调用重定向到选定模型,并保留为请求配置的推理强度。默认源前缀为 `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`;0.144.x 及更早客户端使用 `gpt-5.4-mini`,可通过 `sourceModels` 恢复。 | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | 在可用时启用 | Web 搜索侧车选项。 | | `visionSidecar?` | `OcxVisionSidecarConfig` | 在可用时启用 | 图像描述侧车选项。 | | `images?` | `OcxImagesConfig` | 自动选择 OpenAI | 用于 Codex `image_gen` 的独立 Images 转发选项。 | @@ -133,7 +133,7 @@ Codex 会为标题、提交信息等任务使用较小的辅助模型。启用 "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md b/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md index 432f525737f..8573f495aed 100644 --- a/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md +++ b/docs-site/src/content/docs/zh-tw/reference/cli/providers-accounts.md @@ -303,7 +303,7 @@ Preview 建置使用 `/native-main-profiles`。該配置絕不 | `provider ` | `--json` | 在單次寫入中啟用或停用一個供應商的所有模型。 | | `selected ` | `--set `, `--clear`, `--json` | 讀取或替換供應商模型允許清單。`--clear` 移除允許清單,使每個模型都被提供。 | | `context \|provider \|all >` | `--json` | 讀取或設定 context-window 上限,全域或 per 供應商。 | -| `shadow [model\|-]` | `--enabled `, `--json` | 讀取或設定 Codex 背景 helper 呼叫的替換模型。`-` 清除模型。`status` 亦回報 `sourceModels`,即代理攔截的 helper slug(預設:`gpt-6-luna`, `gpt-5.6-luna`;0.144.x 以前的用戶端使用已退役的 `gpt-5.4-mini`,可透過 `sourceModels` 還原)。 | +| `shadow [model\|-]` | `--enabled `, `--json` | 讀取或設定 Codex 背景 helper 呼叫的替換模型。`-` 清除模型。`status` 亦回報 `sourceModels`,即代理攔截的 helper slug(預設:`gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`;0.144.x 以前的用戶端使用已退役的 `gpt-5.4-mini`,可透過 `sourceModels` 還原)。 | ```bash ocx models live --json # Codex 目前實際可見的模型 diff --git a/docs-site/src/content/docs/zh-tw/reference/configuration/server.md b/docs-site/src/content/docs/zh-tw/reference/configuration/server.md index ec8ffe6e41f..f1d062e6abe 100644 --- a/docs-site/src/content/docs/zh-tw/reference/configuration/server.md +++ b/docs-site/src/content/docs/zh-tw/reference/configuration/server.md @@ -25,7 +25,7 @@ description: 監聽器、遠端存取、許可金鑰、逾時、儲存、sidecar | `codexAutoStart?` | `boolean` | `true` | 讓 Codex shim 在啟動 Codex 前執行 `ocx ensure`。False 使 ensure 為 no-op。 | | `codexShimAutoRestore?` | `boolean` | `true` | 在完成的外部 Codex 更新取代已安裝的 shim 後還原它。環境退出:`OPENCODEX_CODEX_SHIM_AUTO_RESTORE=0`。 | | `syncResumeHistory?` | `boolean` | `true` | 可逆的 Codex App 歷史相容性。原始中繼資料由 `ocx stop` / `ocx restore` 備份並還原。 | -| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | 將識別的 Codex helper/shadow call 重定向到所選模型,並保留為請求設定的 reasoning effort。預設來源前綴為 `gpt-6-luna`, `gpt-5.6-luna`;0.144.x 及更舊的客戶端使用 `gpt-5.4-mini`,可透過 `sourceModels` 恢復。 | +| `shadowCallIntercept?` | `{ enabled?: boolean; model?: string; sourceModels?: string[] }` | off | 將識別的 Codex helper/shadow call 重定向到所選模型,並保留為請求設定的 reasoning effort。預設來源前綴為 `gpt-6-luna`, `gpt-5.6-luna`, `gpt-5.6-terra`;0.144.x 及更舊的客戶端使用 `gpt-5.4-mini`,可透過 `sourceModels` 恢復。 | | `webSearchSidecar?` | `OcxWebSearchSidecarConfig` | 可用時開啟 | 網頁搜尋 sidecar 選項。 | | `visionSidecar?` | `OcxVisionSidecarConfig` | 可用時開啟 | 圖片描述 sidecar 選項。 | | `images?` | `OcxImagesConfig` | 自動 OpenAI 選擇 | Codex `image_gen` 的獨立 Images 中繼選項。 | @@ -155,7 +155,7 @@ Codex 使用小型 helper 模型處理如標題與 commit 訊息等任務。啟 "shadowCallIntercept": { "enabled": true, "model": "gpt-5.5", - "sourceModels": ["gpt-6-luna", "gpt-5.6-luna"] + "sourceModels": ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] } } ``` diff --git a/gui/src/components/MemoryModelsPanel.tsx b/gui/src/components/MemoryModelsPanel.tsx new file mode 100644 index 00000000000..595f166843f --- /dev/null +++ b/gui/src/components/MemoryModelsPanel.tsx @@ -0,0 +1,226 @@ +import { useCallback, useEffect, useRef, useState } from "react"; +import { useT, type TKey } from "../i18n/shared"; +import { IconAlert, IconInfo, IconX } from "../icons"; +import { Select } from "../ui"; +import { createBoundedFetch } from "../bounded-fetch"; +import { requireJson, useModalDialog, type ModelInfo } from "../pages/dashboard-shared"; +import { formatNamespacedModelId } from "../provider-icons"; + +type Phase = "extract" | "consolidation"; +interface PhaseSetting { model?: string; reasoningEffort?: string } +type Settings = { extract?: PhaseSetting; consolidation?: PhaseSetting }; + +const EFFORTS = ["none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra"]; + +/** + * Read the persisted phases. A phase without a model is "Off", so it is dropped rather than kept + * as an empty row: that is also the shape the PUT sends back for it. + */ +function readSettings(payload: { memoryModels?: unknown }): Settings { + const value = payload.memoryModels; + if (value == null) return {}; + if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("invalid settings"); + const out: Settings = {}; + for (const phase of ["extract", "consolidation"] as const) { + const raw = (value as Record)[phase]; + if (raw === undefined) continue; + if (!raw || typeof raw !== "object" || Array.isArray(raw)) throw new Error("invalid phase"); + const model = "model" in raw && typeof raw.model === "string" ? raw.model.trim() : ""; + if (!model) throw new Error("invalid model"); + const effort = "reasoningEffort" in raw ? raw.reasoningEffort : undefined; + if (effort !== undefined && (typeof effort !== "string" || !EFFORTS.includes(effort))) throw new Error("invalid effort"); + out[phase] = { model, ...(effort ? { reasoningEffort: effort } : {}) }; + } + return out; +} + +/** + * One phase's PUT payload. A phase with no model is "Off", which the route reads as an absent + * key, so it must stay out of the object rather than travel as an empty string. + */ +function phasePayload(model: string, effort: string): PhaseSetting | undefined { + return model ? { model, ...(effort ? { reasoningEffort: effort } : {}) } : undefined; +} + +export default function MemoryModelsPanel(props: { apiBase: string; models: ModelInfo[] }) { + return ; +} + +function MemoryModelsControls({ apiBase, models }: { apiBase: string; models: ModelInfo[] }) { + const t = useT(); + const [saved, setSaved] = useState(undefined); + const [infoOpen, setInfoOpen] = useState(false); + const [extractModel, setExtractModel] = useState(""); + const [extractEffort, setExtractEffort] = useState(""); + const [consolidationModel, setConsolidationModel] = useState(""); + const [consolidationEffort, setConsolidationEffort] = useState(""); + const [busy, setBusy] = useState(false); + const [loadError, setLoadError] = useState(false); + const [feedback, setFeedback] = useState<"saved" | "failed" | null>(null); + const active = useRef(false); + const pending = useRef | null>(null); + const infoTriggerRef = useRef(null); + const infoDialogRef = useModalDialog(infoOpen, infoTriggerRef); + + const accept = useCallback((value: Settings) => { + setSaved(value); + setExtractModel(value.extract?.model ?? ""); + setExtractEffort(value.extract?.reasoningEffort ?? ""); + setConsolidationModel(value.consolidation?.model ?? ""); + setConsolidationEffort(value.consolidation?.reasoningEffort ?? ""); + }, []); + + const load = useCallback(async () => { + if (pending.current) return; + const request = createBoundedFetch(15_000); + pending.current = request; + setLoadError(false); + try { + const response = await fetch(`${apiBase}/api/settings`, { signal: request.signal }); + const value = readSettings(await requireJson(response)); + if (active.current && pending.current === request) accept(value); + } catch { + if (active.current && pending.current === request) setLoadError(true); + } finally { + request.clear(); + if (pending.current === request) pending.current = null; + } + }, [apiBase, accept]); + + useEffect(() => { + active.current = true; + const timer = window.setTimeout(() => { void load(); }, 0); + return () => { + window.clearTimeout(timer); + active.current = false; + pending.current?.controller.abort(); + pending.current?.clear(); + pending.current = null; + }; + }, [load]); + + const save = async () => { + if (pending.current || saved === undefined) return; + const request = createBoundedFetch(15_000); + pending.current = request; + setBusy(true); + setFeedback(null); + const extract = phasePayload(extractModel, extractEffort); + const consolidation = phasePayload(consolidationModel, consolidationEffort); + try { + const response = await fetch(`${apiBase}/api/settings`, { + method: "PUT", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + // Null clears the whole block; a phase left at "Off" is simply absent. + memoryModels: extract || consolidation + ? { ...(extract ? { extract } : {}), ...(consolidation ? { consolidation } : {}) } + : null, + }), + signal: request.signal, + }); + const value = readSettings(await requireJson(response)); + if (active.current && pending.current === request) { + accept(value); + setFeedback("saved"); + } + } catch { + if (active.current && pending.current === request) setFeedback("failed"); + } finally { + request.clear(); + if (active.current && pending.current === request) setBusy(false); + if (pending.current === request) pending.current = null; + } + }; + + const options = [{ value: "", label: t("memoryModels.off") }, + ...[...new Set([...models.map(item => item.namespaced), + ...[extractModel, consolidationModel].filter(Boolean)])] + .map(value => ({ value, label: formatNamespacedModelId(value, t) }))]; + const effortOptions = [{ value: "", label: t("memoryModels.defaultEffort") }, + ...EFFORTS.map(value => ({ value, label: t(`models.reasoningEffort.${value}` as TKey) }))]; + const disabled = busy || saved === undefined || loadError; + const dirty = extractModel !== (saved?.extract?.model ?? "") + || extractEffort !== (saved?.extract?.reasoningEffort ?? "") + || consolidationModel !== (saved?.consolidation?.model ?? "") + || consolidationEffort !== (saved?.consolidation?.reasoningEffort ?? ""); + // The account notice is about the phase that stays on Codex's own model, so it is both + // true and useful only while exactly one of the two phases is routed. + const partiallyRouted = Boolean(extractModel) !== Boolean(consolidationModel); + const info = t("memoryModels.info"); + + const row = (phase: Phase, model: string, effort: string, setModel: (value: string) => void, setEffort: (value: string) => void) => ( +
+
+
{t(`memoryModels.${phase}` as TKey)}
+
{t(`memoryModels.${phase}Hint` as TKey)}
+
+
+ { setEffort(value); setFeedback(null); }} /> +
+
+ ); + + return ( +
+
+ {t("memoryModels.title")} + +
+
{t("memoryModels.description")}
+ {row("extract", extractModel, extractEffort, setExtractModel, setExtractEffort)} + {row("consolidation", consolidationModel, consolidationEffort, setConsolidationModel, setConsolidationEffort)} +
+
{t("memoryModels.dataNotice")}
+ +
+ {partiallyRouted &&
+ {t("memoryModels.accountNotice")} +
} + {loadError &&
{t("memoryModels.loadFailed")}
} + {feedback === "failed" &&
{t("memoryModels.saveFailed")}
} + {feedback === "saved" &&
{t("memoryModels.saved")}
} + { event.preventDefault(); setInfoOpen(false); }} + > + + +
+ {info} +
+
+ +
+ +
+
+ ); +} diff --git a/gui/src/i18n/de.ts b/gui/src/i18n/de.ts index c9dfd974a3c..9ea23299c66 100644 --- a/gui/src/i18n/de.ts +++ b/gui/src/i18n/de.ts @@ -422,6 +422,23 @@ export const de: Record = { "compactionRouting.loadFailed": "Komprimierungseinstellungen konnten nicht geladen werden.", "compactionRouting.saved": "Komprimierungseinstellungen gespeichert.", "compactionRouting.saveFailed": "Speichern fehlgeschlagen. Deine Änderungen sind noch vorhanden; versuche es erneut.", + "memoryModels.title": "Memory-Routing", + "memoryModels.description": "Codex schreibt Memories im Hintergrund, nachdem eine Sitzung endet. Wähle das Modell für jeden Schritt — oder lass Codex selbst wählen.", + "memoryModels.infoLabel": "Was sind Extract und Consolidation?", + "memoryModels.info": "Codex macht Memory in zwei Schritten. Extract liest eine beendete Sitzung und notiert, was passiert ist: ein Notizzettel pro Sitzung, also viele kleine Aufrufe. Consolidation nimmt diese Zettel und schreibt sie in die Memory-Dateien, die Codex am Anfang deiner nächsten Sitzungen liest. Läuft selten, bearbeitet aber Dateien. Jeder Schritt fragt sein Modell selbst an, deshalb stehen sie hier getrennt.", + "memoryModels.extract": "Extraktion", + "memoryModels.extractHint": "Fasst jede beendete Sitzung zu einem Raw Memory zusammen. Läuft einmal pro Sitzung.", + "memoryModels.consolidation": "Konsolidierung", + "memoryModels.consolidationHint": "Führt die Raw Memories in die Memory-Dateien zusammen, die Codex später liest. Läuft selten und bearbeitet Dateien.", + "memoryModels.model": "Modell", + "memoryModels.effort": "Reasoning-Aufwand", + "memoryModels.off": "Aus — Codex-Standard", + "memoryModels.defaultEffort": "Codex-Standard", + "memoryModels.dataNotice": "Das gewählte Modell erhält die Eingabe seiner Phase: die beendete Sitzung bei Extract, die Raw Memories bei Consolidation.", + "memoryModels.accountNotice": "Nur eine Phase ist hier geroutet; die andere behält ihre bestehende Route. Der Shadow Call Intercept kann auch die Memory-Aufrufe dieser Phase an sein eingestelltes Modell schicken.", + "memoryModels.loadFailed": "Memory-Einstellungen konnten nicht geladen werden.", + "memoryModels.saved": "Memory-Einstellungen gespeichert.", + "memoryModels.saveFailed": "Speichern fehlgeschlagen. Deine Änderungen stehen noch da; versuch es erneut.", "dash.shadowCallIntercept": "Shadow-Call-Abfangen", "dash.shadowCallInterceptHint": "Fängt die Hintergrund-Hilfsaufrufe der Codex-App ({models}) ab und leitet sie an das gewählte Modell um.", "dash.shadowCallWarning": "⚠ Bei Aktivierung werden ALLE Anfragen an {models} durch das gewählte Modell ersetzt.", diff --git a/gui/src/i18n/en.ts b/gui/src/i18n/en.ts index d5eeb2e9273..761d3f881dd 100644 --- a/gui/src/i18n/en.ts +++ b/gui/src/i18n/en.ts @@ -440,6 +440,23 @@ export const en = { "compactionRouting.loadFailed": "Could not load compaction settings.", "compactionRouting.saved": "Compaction settings saved.", "compactionRouting.saveFailed": "Could not save. Your changes are still here; try again.", + "memoryModels.title": "Memory routing", + "memoryModels.description": "Codex writes memories in the background after a session ends. Pick the model each step uses, or leave it on Codex's own choice.", + "memoryModels.infoLabel": "What are Extract and Consolidation?", + "memoryModels.info": "Codex turns finished sessions into memory in two steps. Extract reads one finished session and jots down what happened: one note per session, so it makes many small calls. Consolidation takes those notes and writes them into the memory files Codex reads at the start of your next sessions. It runs rarely, but it edits files. Each step asks for its own model, which is why they are listed separately here.", + "memoryModels.extract": "Extract", + "memoryModels.extractHint": "Summarizes each finished session into a raw memory. Runs once per session.", + "memoryModels.consolidation": "Consolidation", + "memoryModels.consolidationHint": "Merges the raw memories into the memory files Codex reads later. Runs rarely and edits files.", + "memoryModels.model": "Model", + "memoryModels.effort": "Reasoning effort", + "memoryModels.off": "Off — Codex default", + "memoryModels.defaultEffort": "Codex default", + "memoryModels.dataNotice": "The chosen model receives that phase's input: the finished session for Extract, the raw memories for Consolidation.", + "memoryModels.accountNotice": "Only one phase is routed here; the other keeps its existing route. Shadow Call Intercept can also send that phase's memory calls to its configured model.", + "memoryModels.loadFailed": "Could not load memory settings.", + "memoryModels.saved": "Memory settings saved.", + "memoryModels.saveFailed": "Could not save. Your changes are still here; try again.", "dash.shadowCallIntercept": "Shadow Call Intercept", "dash.shadowCallInterceptHint": "Intercepts Codex App's background helper calls ({models}) for title generation and commit messages and redirects them to your chosen model.", "dash.shadowCallWarning": "⚠ When enabled, ALL requests for {models} will be replaced with the selected model.", diff --git a/gui/src/i18n/fr.ts b/gui/src/i18n/fr.ts index 5ca92bf3726..fc76a6faeb8 100644 --- a/gui/src/i18n/fr.ts +++ b/gui/src/i18n/fr.ts @@ -430,6 +430,23 @@ export const fr: Record = { "compactionRouting.loadFailed": "Impossible de charger les paramètres de compaction.", "compactionRouting.saved": "Paramètres de compaction enregistrés.", "compactionRouting.saveFailed": "Échec de l’enregistrement. Vos modifications sont conservées ; réessayez.", + "memoryModels.title": "Routage de la mémoire", + "memoryModels.description": "Codex écrit les mémoires en arrière-plan, une fois la session terminée. Choisissez le modèle de chaque étape, ou laissez Codex décider.", + "memoryModels.infoLabel": "Que sont Extract et Consolidation ?", + "memoryModels.info": "Codex transforme les sessions terminées en mémoire en deux étapes. Extract lit une session terminée et note ce qui s'y est passé : une fiche par session, donc beaucoup de petits appels. Consolidation reprend ces fiches et les écrit dans les fichiers de mémoire que Codex lit au début de vos sessions suivantes. Elle passe rarement, mais elle modifie des fichiers. Chaque étape demande son propre modèle, d'où ces deux lignes.", + "memoryModels.extract": "Extraction", + "memoryModels.extractHint": "Résume chaque session terminée en une mémoire brute. Une fois par session.", + "memoryModels.consolidation": "Consolidation", + "memoryModels.consolidationHint": "Fusionne les mémoires brutes dans les fichiers que Codex lit ensuite. Passe rarement et modifie des fichiers.", + "memoryModels.model": "Modèle", + "memoryModels.effort": "Effort de raisonnement", + "memoryModels.off": "Désactivé — valeur Codex", + "memoryModels.defaultEffort": "Valeur Codex", + "memoryModels.dataNotice": "Le modèle choisi reçoit l'entrée de sa phase : la session terminée pour Extract, les mémoires brutes pour Consolidation.", + "memoryModels.accountNotice": "Une seule phase est routée ici ; l'autre garde sa route actuelle. Shadow Call Intercept peut aussi envoyer les appels de mémoire de cette phase vers son modèle configuré.", + "memoryModels.loadFailed": "Impossible de charger les réglages de mémoire.", + "memoryModels.saved": "Réglages de mémoire enregistrés.", + "memoryModels.saveFailed": "Échec de l'enregistrement. Vos modifications sont conservées ; réessayez.", "dash.shadowCallIntercept": "Interception des appels fantômes", "dash.shadowCallInterceptHint": "Intercepte les appels auxiliaires en arrière-plan de l’application Codex ({models}) pour générer les titres et les messages de commit, puis les redirige vers le modèle choisi.", "dash.shadowCallWarning": "⚠ Lorsque cette option est activée, TOUTES les requêtes destinées à {models} sont remplacées par le modèle sélectionné.", diff --git a/gui/src/i18n/ja.ts b/gui/src/i18n/ja.ts index 67b3a4973f3..787425b03bf 100644 --- a/gui/src/i18n/ja.ts +++ b/gui/src/i18n/ja.ts @@ -431,6 +431,23 @@ export const ja: Record = { "compactionRouting.loadFailed": "圧縮設定を読み込めませんでした。", "compactionRouting.saved": "圧縮設定を保存しました。", "compactionRouting.saveFailed": "保存できませんでした。変更内容は保持されています。再試行してください。", + "memoryModels.title": "メモリルーティング", + "memoryModels.description": "Codex はセッション終了後にバックグラウンドでメモリを書き込みます。各段階で使うモデルを選ぶか、Codex の既定のままにします。", + "memoryModels.infoLabel": "Extract と Consolidation とは?", + "memoryModels.info": "Codex は終了したセッションを 2 段階でメモリにします。Extract は終了したセッションを 1 つ読み、起きたことを書き留めます。セッションごとにメモ 1 枚、つまり小さな呼び出しがたくさん発生します。Consolidation はそのメモをまとめ、次のセッションの開始時に Codex が読むメモリファイルへ書き込みます。めったに動きませんが、ファイルを編集します。各段階が自分のモデルを要求するため、ここでは別々に表示しています。", + "memoryModels.extract": "抽出", + "memoryModels.extractHint": "終了したセッションごとに生のメモリへ要約します。セッションごとに 1 回動きます。", + "memoryModels.consolidation": "統合", + "memoryModels.consolidationHint": "生のメモリを、Codex が後で読むメモリファイルへ統合します。めったに動かず、ファイルを編集します。", + "memoryModels.model": "モデル", + "memoryModels.effort": "推論の強さ", + "memoryModels.off": "オフ — Codex の既定", + "memoryModels.defaultEffort": "Codex の既定", + "memoryModels.dataNotice": "選んだモデルにはその段階の入力が送られます。Extract では終了したセッション、Consolidation では生のメモリです。", + "memoryModels.accountNotice": "ここでは片方の段階だけをルーティングしています。もう一方は既存のルートのままです。Shadow Call Intercept がその段階のメモリ呼び出しを設定済みのモデルへ送ることもあります。", + "memoryModels.loadFailed": "メモリ設定を読み込めませんでした。", + "memoryModels.saved": "メモリ設定を保存しました。", + "memoryModels.saveFailed": "保存できませんでした。変更は残っています。もう一度お試しください。", "dash.shadowCallIntercept": "シャドウコール傍受", "dash.shadowCallInterceptHint": "Codex App のバックグラウンドヘルパー呼び出し({models}: タイトル生成、コミットメッセージ)を傍受し、選択したモデルにリダイレクトします。", "dash.shadowCallWarning": "⚠ オンにすると、{models} へのリクエストがすべて選択したモデルに置き換えられます。", diff --git a/gui/src/i18n/ko.ts b/gui/src/i18n/ko.ts index ecb3c2b1003..be290d764b1 100644 --- a/gui/src/i18n/ko.ts +++ b/gui/src/i18n/ko.ts @@ -426,6 +426,23 @@ export const ko: Record = { "compactionRouting.loadFailed": "압축 설정을 불러올 수 없습니다.", "compactionRouting.saved": "압축 설정을 저장했습니다.", "compactionRouting.saveFailed": "저장하지 못했습니다. 변경 사항은 유지됩니다. 다시 시도하세요.", + "memoryModels.title": "메모리 라우팅", + "memoryModels.description": "Codex는 세션이 끝난 뒤 백그라운드에서 메모리를 작성합니다. 각 단계에 쓸 모델을 고르거나 Codex의 기본 선택을 그대로 두세요.", + "memoryModels.infoLabel": "Extract와 Consolidation이 무엇인가요?", + "memoryModels.info": "Codex는 끝난 세션을 두 단계로 메모리로 만듭니다. Extract는 끝난 세션 하나를 읽고 무슨 일이 있었는지 적습니다. 세션마다 메모 한 장, 즉 작은 호출이 많습니다. Consolidation은 그 메모를 모아 다음 세션 시작에 Codex가 읽는 메모리 파일에 씁니다. 드물게 실행되지만 파일을 수정합니다. 각 단계가 자기 모델을 요청하기 때문에 여기에 따로 표시됩니다.", + "memoryModels.extract": "추출", + "memoryModels.extractHint": "끝난 세션마다 원시 메모리로 요약합니다. 세션당 한 번 실행됩니다.", + "memoryModels.consolidation": "통합", + "memoryModels.consolidationHint": "원시 메모리를 Codex가 나중에 읽는 메모리 파일로 합칩니다. 드물게 실행되며 파일을 수정합니다.", + "memoryModels.model": "모델", + "memoryModels.effort": "추론 노력", + "memoryModels.off": "사용 안 함 — Codex 기본값", + "memoryModels.defaultEffort": "Codex 기본값", + "memoryModels.dataNotice": "선택한 모델은 해당 단계의 입력을 받습니다. Extract는 끝난 세션, Consolidation은 원시 메모리입니다.", + "memoryModels.accountNotice": "여기서는 한 단계만 라우팅했습니다. 나머지 단계는 기존 경로를 그대로 씁니다. Shadow Call Intercept가 그 단계의 메모리 호출을 설정된 모델로 보낼 수도 있습니다.", + "memoryModels.loadFailed": "메모리 설정을 불러오지 못했습니다.", + "memoryModels.saved": "메모리 설정을 저장했습니다.", + "memoryModels.saveFailed": "저장하지 못했습니다. 변경 사항은 그대로 있습니다. 다시 시도하세요.", "dash.shadowCallIntercept": "쉐도우 호출 가로채기", "dash.shadowCallInterceptHint": "Codex 앱이 제목·커밋 메시지 생성에 쓰는 백그라운드 호출({models})을 가로채 선택한 모델로 바꿉니다.", "dash.shadowCallWarning": "⚠ 활성화하면 {models} 요청이 모두 선택한 모델로 대체됩니다.", diff --git a/gui/src/i18n/ru.ts b/gui/src/i18n/ru.ts index 16a41bf6975..c16ca791025 100644 --- a/gui/src/i18n/ru.ts +++ b/gui/src/i18n/ru.ts @@ -431,6 +431,23 @@ export const ru: Record = { "compactionRouting.loadFailed": "Не удалось загрузить настройки сжатия.", "compactionRouting.saved": "Настройки сжатия сохранены.", "compactionRouting.saveFailed": "Не удалось сохранить. Изменения остались; попробуйте снова.", + "memoryModels.title": "Маршрутизация памяти", + "memoryModels.description": "Codex пишет память в фоне после завершения сессии. Выберите модель для каждого шага или оставьте выбор Codex.", + "memoryModels.infoLabel": "Что такое Extract и Consolidation?", + "memoryModels.info": "Codex превращает завершённые сессии в память за два шага. Extract читает одну завершённую сессию и записывает, что в ней произошло: одна заметка на сессию, то есть много небольших вызовов. Consolidation берёт эти заметки и записывает их в файлы памяти, которые Codex читает в начале следующих сессий. Запускается редко, но изменяет файлы. Каждый шаг сам запрашивает модель, поэтому они показаны отдельно.", + "memoryModels.extract": "Извлечение", + "memoryModels.extractHint": "Сводит каждую завершённую сессию в одну сырую запись. Запускается раз на сессию.", + "memoryModels.consolidation": "Консолидация", + "memoryModels.consolidationHint": "Сводит сырые записи в файлы памяти, которые Codex читает позже. Запускается редко и изменяет файлы.", + "memoryModels.model": "Модель", + "memoryModels.effort": "Усилие рассуждений", + "memoryModels.off": "Выключено — по умолчанию Codex", + "memoryModels.defaultEffort": "Как в Codex", + "memoryModels.dataNotice": "Выбранная модель получает входные данные своей фазы: завершённую сессию для Extract и сырые записи для Consolidation.", + "memoryModels.accountNotice": "Здесь смаршрутизирована только одна фаза; другая сохраняет свой текущий маршрут. Shadow Call Intercept тоже может отправлять вызовы памяти этой фазы в свою настроенную модель.", + "memoryModels.loadFailed": "Не удалось загрузить настройки памяти.", + "memoryModels.saved": "Настройки памяти сохранены.", + "memoryModels.saveFailed": "Не удалось сохранить. Ваши изменения на месте; попробуйте снова.", "dash.shadowCallIntercept": "Перехват теневых вызовов", "dash.shadowCallInterceptHint": "Перехватывает фоновые служебные вызовы Codex App ({models}: генерация заголовков, сообщений коммитов) и перенаправляет их на выбранную вами модель.", "dash.shadowCallWarning": "⚠ Когда функция включена, ВСЕ запросы к {models} будут заменены выбранной моделью.", diff --git a/gui/src/i18n/tr.ts b/gui/src/i18n/tr.ts index b592e84752c..c034e445a46 100644 --- a/gui/src/i18n/tr.ts +++ b/gui/src/i18n/tr.ts @@ -432,6 +432,23 @@ export const tr: Record = { "compactionRouting.loadFailed": "Özetleme ayarları yüklenemedi.", "compactionRouting.saved": "Özetleme ayarları kaydedildi.", "compactionRouting.saveFailed": "Kaydedilemedi. Değişiklikleriniz korunuyor; tekrar deneyin.", + "memoryModels.title": "Bellek yönlendirmesi", + "memoryModels.description": "Codex, oturum bittikten sonra belleği arka planda yazar. Her adım için modeli siz seçin ya da Codex'in kendi seçiminde bırakın.", + "memoryModels.infoLabel": "Extract ve Consolidation nedir?", + "memoryModels.info": "Codex, biten oturumları iki adımda belleğe dönüştürür. Extract biten bir oturumu okuyup ne olduğunu not eder: oturum başına bir not, yani çok sayıda küçük çağrı. Consolidation bu notları alıp Codex'in sonraki oturumların başında okuduğu bellek dosyalarına yazar. Seyrek çalışır ama dosyaları düzenler. Her adım kendi modelini ister, bu yüzden burada ayrı görünürler.", + "memoryModels.extract": "Çıkarım", + "memoryModels.extractHint": "Her biten oturumu bir ham belleğe özetler. Oturum başına bir kez çalışır.", + "memoryModels.consolidation": "Birleştirme", + "memoryModels.consolidationHint": "Ham bellekleri Codex'in sonra okuduğu bellek dosyalarında birleştirir. Seyrek çalışır ve dosyaları düzenler.", + "memoryModels.model": "Model", + "memoryModels.effort": "Akıl yürütme çabası", + "memoryModels.off": "Kapalı — Codex varsayılanı", + "memoryModels.defaultEffort": "Codex varsayılanı", + "memoryModels.dataNotice": "Seçilen model kendi aşamasının girdisini alır: Extract için biten oturum, Consolidation için ham bellekler.", + "memoryModels.accountNotice": "Burada yalnızca bir aşama yönlendirildi; diğeri mevcut yolunu korur. Shadow Call Intercept o aşamanın bellek çağrılarını da kendi ayarlı modeline gönderebilir.", + "memoryModels.loadFailed": "Bellek ayarları yüklenemedi.", + "memoryModels.saved": "Bellek ayarları kaydedildi.", + "memoryModels.saveFailed": "Kaydedilemedi. Değişiklikleriniz duruyor; tekrar deneyin.", "dash.shadowCallIntercept": "Gölge Çağrı Yakalama", "dash.shadowCallInterceptHint": "Codex App'in arka plan yardımcı çağrılarını ({models}) başlık oluşturma ve commit mesajları için yakalar ve seçtiğiniz modele yönlendirir.", "dash.shadowCallWarning": "⚠ Etkinleştirildiğinde, {models} için olan TÜM istekler seçilen modelle değiştirilecektir.", diff --git a/gui/src/i18n/vi.ts b/gui/src/i18n/vi.ts index 4be42b9ebc6..8a4552ab23c 100644 --- a/gui/src/i18n/vi.ts +++ b/gui/src/i18n/vi.ts @@ -430,6 +430,23 @@ export const vi: Record = { "compactionRouting.loadFailed": "Không thể tải cài đặt nén.", "compactionRouting.saved": "Đã lưu cài đặt nén.", "compactionRouting.saveFailed": "Không thể lưu. Thay đổi của bạn vẫn còn; hãy thử lại.", + "memoryModels.title": "Định tuyến bộ nhớ", + "memoryModels.description": "Codex ghi bộ nhớ ở chế độ nền sau khi phiên kết thúc. Chọn mô hình cho từng bước, hoặc để Codex tự chọn.", + "memoryModels.infoLabel": "Extract và Consolidation là gì?", + "memoryModels.info": "Codex biến các phiên đã kết thúc thành bộ nhớ qua hai bước. Extract đọc một phiên đã kết thúc và ghi lại những gì đã diễn ra: một ghi chú cho mỗi phiên, nên có nhiều lời gọi nhỏ. Consolidation lấy các ghi chú đó và ghi vào các tệp bộ nhớ mà Codex đọc khi bắt đầu các phiên sau. Hiếm khi chạy nhưng có sửa tệp. Mỗi bước tự yêu cầu mô hình riêng, nên ở đây chúng được tách riêng.", + "memoryModels.extract": "Trích xuất", + "memoryModels.extractHint": "Tóm tắt mỗi phiên đã kết thúc thành một bộ nhớ thô. Chạy một lần mỗi phiên.", + "memoryModels.consolidation": "Hợp nhất", + "memoryModels.consolidationHint": "Hợp nhất các bộ nhớ thô vào tệp bộ nhớ mà Codex đọc sau này. Hiếm khi chạy và có sửa tệp.", + "memoryModels.model": "Mô hình", + "memoryModels.effort": "Mức suy luận", + "memoryModels.off": "Tắt — mặc định của Codex", + "memoryModels.defaultEffort": "Mặc định của Codex", + "memoryModels.dataNotice": "Mô hình được chọn nhận đầu vào của giai đoạn đó: phiên đã kết thúc với Extract, bộ nhớ thô với Consolidation.", + "memoryModels.accountNotice": "Ở đây chỉ có một giai đoạn được định tuyến; giai đoạn còn lại giữ nguyên tuyến hiện có. Shadow Call Intercept cũng có thể gửi các lệnh gọi bộ nhớ của giai đoạn đó tới mô hình đã đặt.", + "memoryModels.loadFailed": "Không tải được cài đặt bộ nhớ.", + "memoryModels.saved": "Đã lưu cài đặt bộ nhớ.", + "memoryModels.saveFailed": "Không lưu được. Thay đổi của bạn vẫn còn; hãy thử lại.", "dash.shadowCallIntercept": "Shadow Call Intercept", "dash.shadowCallInterceptHint": "Chặn các lệnh gọi helper nền ({models}) của ứng dụng Codex để tạo tiêu đề và commit messages, sau đó chuyển hướng chúng đến model bạn đã chọn.", "dash.shadowCallWarning": "⚠ Khi được bật, TẤT CẢ yêu cầu đối với {models} sẽ được thay thế bằng model được chọn.", diff --git a/gui/src/i18n/zh-TW.ts b/gui/src/i18n/zh-TW.ts index 9285fe2a451..d9d150bd9aa 100644 --- a/gui/src/i18n/zh-TW.ts +++ b/gui/src/i18n/zh-TW.ts @@ -308,6 +308,23 @@ export const zhTW: Record = { "compactionRouting.loadFailed": "無法載入壓縮設定。", "compactionRouting.saved": "壓縮設定已儲存。", "compactionRouting.saveFailed": "儲存失敗。變更仍然保留,請重試。", + "memoryModels.title": "記憶路由", + "memoryModels.description": "工作階段結束後,Codex 會在背景寫入記憶。為每個步驟選擇模型,或保留 Codex 自己的選擇。", + "memoryModels.infoLabel": "Extract 和 Consolidation 是什麼?", + "memoryModels.info": "Codex 用兩個步驟把結束的工作階段變成記憶。Extract 讀取一個已結束的工作階段並記下其中發生的事:每個工作階段一張筆記,因此會有很多小型請求。Consolidation 把這些筆記寫進 Codex 在後續工作階段開始時讀取的記憶檔案。很少執行,但會修改檔案。兩個步驟各自要求自己的模型,所以這裡分開顯示。", + "memoryModels.extract": "擷取", + "memoryModels.extractHint": "把每個結束的工作階段彙整成一筆原始記憶。每個工作階段執行一次。", + "memoryModels.consolidation": "合併", + "memoryModels.consolidationHint": "把原始記憶合併進 Codex 之後讀取的記憶檔案。很少執行,而且會修改檔案。", + "memoryModels.model": "模型", + "memoryModels.effort": "推理強度", + "memoryModels.off": "關閉 — Codex 預設", + "memoryModels.defaultEffort": "Codex 預設", + "memoryModels.dataNotice": "所選模型會收到該階段的輸入:Extract 是已結束的工作階段,Consolidation 是原始記憶。", + "memoryModels.accountNotice": "這裡只路由了一個階段;另一個階段保留其現有路由。Shadow Call Intercept 也可能把該階段的記憶呼叫送往其設定的模型。", + "memoryModels.loadFailed": "無法載入記憶設定。", + "memoryModels.saved": "記憶設定已儲存。", + "memoryModels.saveFailed": "儲存失敗。你的變更還在,請再試一次。", "dash.shadowCallIntercept": "影子呼叫攔截", "dash.shadowCallInterceptHint": "攔截 Codex 應用的背景 helper 呼叫({models})以生成標題與提交訊息,並將它們重定向到您選擇的模型。", "dash.shadowCallWarning": "⚠ 啟用後,{models} 的所有請求將被替換為所選模型。", diff --git a/gui/src/i18n/zh.ts b/gui/src/i18n/zh.ts index d30a30fe4e1..bf359d82ebc 100644 --- a/gui/src/i18n/zh.ts +++ b/gui/src/i18n/zh.ts @@ -426,6 +426,23 @@ export const zh: Record = { "compactionRouting.loadFailed": "无法加载压缩设置。", "compactionRouting.saved": "压缩设置已保存。", "compactionRouting.saveFailed": "保存失败。更改仍然保留,请重试。", + "memoryModels.title": "记忆路由", + "memoryModels.description": "会话结束后,Codex 会在后台写入记忆。为每个步骤选择模型,或保留 Codex 自己的选择。", + "memoryModels.infoLabel": "Extract 和 Consolidation 是什么?", + "memoryModels.info": "Codex 分两步把结束的会话变成记忆。Extract 读取一个已结束的会话并记下其中的内容:每个会话一张笔记,因此有很多小请求。Consolidation 把这些笔记写进 Codex 在后续会话开始时读取的记忆文件。很少运行,但会修改文件。两步各自请求自己的模型,所以这里分开展示。", + "memoryModels.extract": "提取", + "memoryModels.extractHint": "把每个结束的会话汇总成一条原始记忆。每个会话运行一次。", + "memoryModels.consolidation": "整合", + "memoryModels.consolidationHint": "把原始记忆合并进 Codex 之后读取的记忆文件。很少运行,并会修改文件。", + "memoryModels.model": "模型", + "memoryModels.effort": "推理强度", + "memoryModels.off": "关闭 — Codex 默认", + "memoryModels.defaultEffort": "Codex 默认", + "memoryModels.dataNotice": "所选模型会收到该阶段的输入:Extract 是已结束的会话,Consolidation 是原始记忆。", + "memoryModels.accountNotice": "这里只路由了一个阶段;另一个阶段保留其现有路由。Shadow Call Intercept 也可能把该阶段的记忆调用发往其配置的模型。", + "memoryModels.loadFailed": "无法加载记忆设置。", + "memoryModels.saved": "记忆设置已保存。", + "memoryModels.saveFailed": "保存失败。你的改动仍在,请重试。", "dash.shadowCallIntercept": "影子调用拦截", "dash.shadowCallInterceptHint": "拦截 Codex 应用的后台辅助调用({models}:标题生成、提交消息)并重定向到所选模型。", "dash.shadowCallWarning": "⚠ 启用后,所有对 {models} 的请求都将被替换为所选模型。", diff --git a/gui/src/pages/dashboard-overview-panels.tsx b/gui/src/pages/dashboard-overview-panels.tsx index 9d6a9e43684..afe2874d380 100644 --- a/gui/src/pages/dashboard-overview-panels.tsx +++ b/gui/src/pages/dashboard-overview-panels.tsx @@ -1,5 +1,6 @@ import CompactionRoutingPanel from "../components/CompactionRoutingPanel"; import MemoryObservabilityCard from "../components/MemoryObservabilityCard"; +import MemoryModelsPanel from "../components/MemoryModelsPanel"; import type { useDashboardData } from "./use-dashboard-data"; import { DashboardEffortCapPanel, @@ -20,6 +21,7 @@ export function DashboardOverviewPanels(props: Dash) { + ); diff --git a/gui/src/pages/shadow-call-source.ts b/gui/src/pages/shadow-call-source.ts index a75fcb47443..bb4e1b062ba 100644 --- a/gui/src/pages/shadow-call-source.ts +++ b/gui/src/pages/shadow-call-source.ts @@ -2,12 +2,13 @@ * Which models the runtime actually intercepts as shadow calls. * * Codex 0.154.0+ sends gpt-6-luna for helper calls; 0.145.0-0.153.x sent - * gpt-5.6-luna, which the runtime still intercepts by default. Clients through + * gpt-5.6-luna, which the runtime still intercepts by default, and + * gpt-5.6-terra is the background memory-consolidation model. Clients through * 0.144.x used gpt-5.4-mini, which operators can restore through `sourceModels`. * The GUI renders whatever the runtime reports rather than a baked-in label; * this fallback only covers a runtime too old to send `sourceModels`. */ -const FALLBACK_SOURCE_MODELS = ["gpt-6-luna", "gpt-5.6-luna"]; +const FALLBACK_SOURCE_MODELS = ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"]; export function shadowSourceModelList(sourceModels?: string[]): string[] { const cleaned = Array.isArray(sourceModels) diff --git a/gui/src/styles-dashboard-workspace.css b/gui/src/styles-dashboard-workspace.css index 22f60ef9159..71b08d4e470 100644 --- a/gui/src/styles-dashboard-workspace.css +++ b/gui/src/styles-dashboard-workspace.css @@ -655,6 +655,52 @@ .dash-shadow-controls > .custom-select { flex: 1; min-width: 0; } .dash-shadow-controls > .switch { order: 1; } +/* Memory routing: two fixed column widths, shared by both rows. The pickers must not + size themselves from their own label — content-sized pills turned "Low" and "Medium" + into two different widths, so the Consolidation row never lined up with the Extract + row above it, and the effort picker alone wasted 220px on a five-letter word. + + Each column gets its own width and both rows use the same two, so the columns compare + across the rows instead of inside one row: the model column is exactly as wide as the + longest common id needs (measured 220px at the trigger's 13px font, trigger padding, + chevron gap and chevron included), the effort column fits the longest option label + "Codex default" (measured 136px, rounded to 140px). A longer model id ellipsizes + inside its column instead of widening one row's pill, so the two rows keep matching + columns whatever is selected. */ +.memory-models-row { align-items: center; } + +.memory-models-controls { + display: flex; + align-items: center; + flex-wrap: nowrap; + gap: 8px; + flex: 0 0 min(100%, 23rem); + min-width: 0; +} + +.memory-models-controls > .custom-select:first-child { + flex: 0 1 220px; + min-width: 0; +} + +.memory-models-controls > .custom-select:last-child { + flex: 0 1 140px; + min-width: 0; +} + +.memory-models-controls .select-trigger { + justify-content: space-between; + width: 100%; + min-width: 0; +} + +.memory-models-controls .select-trigger > span { + min-width: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + .dash-overview-stack .setting-hint, .dash-overview-stack .dash-sync-hint { max-width: 60ch; } diff --git a/gui/tests/fr-localization.test.ts b/gui/tests/fr-localization.test.ts index 9baa172e2ec..0a2ebb31ecf 100644 --- a/gui/tests/fr-localization.test.ts +++ b/gui/tests/fr-localization.test.ts @@ -218,6 +218,10 @@ const INTENTIONAL_ENGLISH = new Set([ "logs.protocol.wire.chat", "logs.protocol.wire.messages", "logs.protocol.hop.ir", + // The consolidation phase's name is the ordinary French noun, spelled exactly as in English. + // Inventing a synonym would also break the pair with the extract row, whose French label is + // "Extraction". + "memoryModels.consolidation", ]); function placeholders(value: string): string[] { diff --git a/gui/tests/memory-models-panel.test.tsx b/gui/tests/memory-models-panel.test.tsx new file mode 100644 index 00000000000..410935d3369 --- /dev/null +++ b/gui/tests/memory-models-panel.test.tsx @@ -0,0 +1,111 @@ +/** @jsxImportSource react */ +import { afterEach, beforeEach, expect, test } from "bun:test"; +import { Window } from "happy-dom"; +import { act, StrictMode } from "react"; +import type { Root } from "react-dom/client"; +import { LanguageProvider } from "../src/i18n/provider"; +import MemoryModelsPanel from "../src/components/MemoryModelsPanel"; + +const globals = ["document", "window", "navigator", "localStorage", "sessionStorage", "fetch", "HTMLElement", "IS_REACT_ACT_ENVIRONMENT"] as const; +let previous: Record; +let win: Window; +let root: Root | undefined; +let container: HTMLDivElement; +let setting: { extract?: { model: string; reasoningEffort?: string }; consolidation?: { model: string; reasoningEffort?: string } } | null; +let failLoad: boolean; +let failSave: boolean; +let writes: unknown[]; +const models = [{ id: "cheap", provider: "gateway", namespaced: "gateway/cheap" }, { id: "brisk", provider: "combo", namespaced: "combo/brisk" }]; + +beforeEach(() => { + previous = Object.fromEntries(globals.map(key => [key, Object.getOwnPropertyDescriptor(globalThis, key)])); + win = new Window({ url: "http://localhost/" }); + for (const key of ["document", "window", "navigator", "localStorage", "sessionStorage", "HTMLElement"] as const) { + Object.defineProperty(globalThis, key, { configurable: true, value: key === "window" ? win : win[key] }); + } + Object.defineProperty(globalThis, "IS_REACT_ACT_ENVIRONMENT", { configurable: true, value: true }); + win.localStorage.setItem("ocx-lang", "en"); + setting = null; failLoad = false; failSave = false; writes = []; + Object.defineProperty(globalThis, "fetch", { configurable: true, writable: true, value: async (_input: unknown, init?: RequestInit) => { + if (init?.method === "PUT") { + const body = JSON.parse(String(init.body)); + writes.push(body); + if (failSave) return Response.json({ error: "fixture failure" }, { status: 500 }); + setting = body.memoryModels; + } else if (failLoad) return Response.json({ error: "unavailable" }, { status: 503 }); + return Response.json({ memoryModels: setting }); + } }); +}); + +afterEach(async () => { + if (root) await act(async () => { root!.unmount(); }); + root = undefined; win.close(); + for (const key of globals) { + if (previous[key]) Object.defineProperty(globalThis, key, previous[key]!); + else delete (globalThis as Record)[key]; + } +}); + +async function flush() { await act(async () => { await new Promise(resolve => setTimeout(resolve, 10)); }); } +async function render(base = "") { + if (!root) { + container = win.document.createElement("div") as unknown as HTMLDivElement; + win.document.body.appendChild(container); + root = (await import("react-dom/client")).createRoot(container); + } + await act(async () => { root!.render(); }); + await flush(); +} +async function choose(id: string, label: string) { + await act(async () => { container.querySelector("#memory-models-" + id)!.click(); }); + const option = [...win.document.querySelectorAll('[role="option"]')].find(node => node.textContent === label); + expect(option).toBeDefined(); + await act(async () => { (option as unknown as HTMLButtonElement).click(); }); +} +function saveButton() { return [...container.querySelectorAll("button")].find(button => button.textContent === "Save")!; } +async function save() { await act(async () => { saveButton().click(); }); } +function notice() { return container.querySelector('[role="note"]'); } +const OFF = "Off \u2014 Codex default"; + +test("the account notice tracks the half-routed state", async () => { + await render(); + expect(notice()).toBeNull(); + await choose("extract", "gateway/cheap"); + expect(notice()).not.toBeNull(); + await choose("consolidation", "gateway/cheap"); + expect(notice()).toBeNull(); + await choose("extract", OFF); + expect(notice()).not.toBeNull(); + await choose("consolidation", OFF); + expect(notice()).toBeNull(); +}); + +test("saves each phase independently and clears effort with its model", async () => { + await render(); + expect(saveButton().disabled).toBe(true); + await choose("extract", "gateway/cheap"); + await choose("extract-effort", "Low"); + await choose("consolidation", "gateway/cheap"); + await save(); + expect(writes.at(-1)).toEqual({ + memoryModels: { extract: { model: "gateway/cheap", reasoningEffort: "low" }, consolidation: { model: "gateway/cheap" } }, + }); + expect(container.querySelector('[role="status"]')?.textContent).toBe("Memory settings saved."); + // The effort picker is armed only while its phase names a model, and clearing the model + // clears the effort with it, so a phase is either fully routed or absent. + await choose("consolidation", OFF); + expect((container.querySelector("#memory-models-consolidation-effort") as HTMLButtonElement)!.disabled).toBe(true); + await choose("consolidation", "gateway/cheap"); + await choose("consolidation-effort", "Medium"); + await save(); + expect(writes.at(-1)).toEqual({ + memoryModels: { extract: { model: "gateway/cheap", reasoningEffort: "low" }, consolidation: { model: "gateway/cheap", reasoningEffort: "medium" } }, + }); + await choose("extract", OFF); + await save(); + expect(writes.at(-1)).toEqual({ memoryModels: { consolidation: { model: "gateway/cheap", reasoningEffort: "medium" } } }); + await choose("consolidation", OFF); + await save(); + expect(writes.at(-1)).toEqual({ memoryModels: null }); +}); + diff --git a/gui/tests/shadow-call-source.test.ts b/gui/tests/shadow-call-source.test.ts index 21e3c085232..80e8d3f3960 100644 --- a/gui/tests/shadow-call-source.test.ts +++ b/gui/tests/shadow-call-source.test.ts @@ -23,14 +23,14 @@ describe("shadowSourceModelList", () => { }); test("falls back when a runtime too old to report sourceModels omits it", () => { - expect(shadowSourceModelList(undefined)).toEqual(["gpt-6-luna", "gpt-5.6-luna"]); + expect(shadowSourceModelList(undefined)).toEqual(["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"]); }); // An empty array is what a runtime sends when every configured entry was // rejected; showing nothing there would read as "nothing is intercepted", // which is the opposite of the truth. test("falls back on an empty list instead of rendering nothing", () => { - expect(shadowSourceModelList([])).toEqual(["gpt-6-luna", "gpt-5.6-luna"]); + expect(shadowSourceModelList([])).toEqual(["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"]); }); test("drops blank entries and trims the rest", () => { @@ -39,7 +39,7 @@ describe("shadowSourceModelList", () => { test("falls back when the field is not an array at all", () => { expect(shadowSourceModelList("gpt-5.6-luna" as unknown as string[])) - .toEqual(["gpt-6-luna", "gpt-5.6-luna"]); + .toEqual(["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"]); }); }); @@ -59,7 +59,7 @@ describe("shadow source model rendering", () => { }); test("both renderings use the fallback when the runtime reported nothing", () => { - expect(shadowSourceModelLabel(undefined)).toBe("gpt-6-luna, gpt-5.6-luna"); - expect(shadowSourceModelBadge(undefined)).toBe("6-luna, 5.6-luna"); + expect(shadowSourceModelLabel(undefined)).toBe("gpt-6-luna, gpt-5.6-luna, gpt-5.6-terra"); + expect(shadowSourceModelBadge(undefined)).toBe("6-luna, 5.6-luna, 5.6-terra"); }); }); diff --git a/src/config/diagnostics.ts b/src/config/diagnostics.ts index d1ab81747c4..4c630cf169f 100644 --- a/src/config/diagnostics.ts +++ b/src/config/diagnostics.ts @@ -64,6 +64,7 @@ import { spendSchema, compactionRoutingSchema, skillsConfigSchema, + memoryModelsSchema, } from "./schema/leaf-validators"; export type ConfigDiagnostics = { @@ -611,6 +612,10 @@ export function validateConfigCandidate(value: unknown): { ok: true; config: Ocx if (compactionRouting !== undefined && !compactionRoutingSchema.safeParse(compactionRouting).success) { return { ok: false, error: "schema_invalid: compactionRouting: requires a nonblank model, an optional valid reasoningEffort, and optional non-repeating triggers drawn from \"manual\" and \"auto\"" }; } + const memoryModels = rawConfigRecord(value)?.memoryModels; + if (memoryModels !== undefined && !memoryModelsSchema.safeParse(memoryModels).success) { + return { ok: false, error: "schema_invalid: memoryModels: requires a nonblank model and an optional declared reasoningEffort per configured phase, and no other fields" }; + } const boundaryError = compactionRecoveryConfigError(value) ?? configReasoningPinsConfigError(value) ?? blankHostnameError(value) ?? claudeSubagentEffortError(value) diff --git a/src/config/load-degrade.ts b/src/config/load-degrade.ts index aa8cc5e94c4..e2e196b8a75 100644 --- a/src/config/load-degrade.ts +++ b/src/config/load-degrade.ts @@ -122,6 +122,35 @@ export function warnDegradedTopLevelOptIns(rawParsed: unknown, validated: OcxCon if (compactionRecoveryConfigError(rawParsed)) console.warn("⚠️ invalid compactionRecovery disabled; the original compaction failure is preserved"); warnDegradedStreamMode(rawParsed, validated); warnDegradedCompactionRouting(rawParsed, validated); + warnDegradedMemoryModels(rawParsed, validated); +} + +/** + * A malformed `memoryModels` phase disables that phase rather than failing the whole schema, so + * say so once: silently keeping whatever route the phase already had — which may be the shadow + * intercept rather than Codex's own model — is the outcome a typo must not produce quietly. + */ +export function warnDegradedMemoryModels(rawParsed: unknown, validated: OcxConfig): void { + if (!rawParsed || typeof rawParsed !== "object") return; + const raw = (rawParsed as Record).memoryModels; + if (raw === undefined) return; + if (validated.memoryModels === undefined || raw === null || typeof raw !== "object" || Array.isArray(raw)) { + console.warn("\u26a0\ufe0f config.json memoryModels is invalid (expected { extract?: { model, reasoningEffort? }, consolidation?: { model, reasoningEffort? } } with a nonblank model and a declared effort per phase) \u2014 the memory pipeline keeps its existing route, which may include shadow-call interception"); + return; + } + // A misspelled phase key is stripped by the permissive load schema, so without this warning it + // disappears silently and the next settings save persists the sanitized map without it. + for (const key of Object.keys(raw as Record)) { + if (key === "extract" || key === "consolidation") continue; + // Redact and JSON-escape the key name: a malformed hand-edit can place a secret in a property + // name, and a control character in one must not be able to forge a log line. + console.warn("\u26a0\ufe0f config.json memoryModels." + JSON.stringify(redactSecretString(key)) + " is not a recognized phase \u2014 ignoring it"); + } + for (const phase of ["extract", "consolidation"] as const) { + if ((raw as Record)[phase] !== undefined && validated.memoryModels[phase] === undefined) { + console.warn("\u26a0\ufe0f config.json memoryModels." + phase + " is invalid (expected { model, reasoningEffort? } with a nonblank model) \u2014 that phase keeps its existing route, which may include shadow-call interception"); + } + } } /** diff --git a/src/config/schema/config-schema.ts b/src/config/schema/config-schema.ts index a855202a29a..238c5647ad2 100644 --- a/src/config/schema/config-schema.ts +++ b/src/config/schema/config-schema.ts @@ -25,6 +25,8 @@ import { codexAccountNamespacesSchema, modelPinnedEffortsSchema, compactionRoutingSchema, + memoryModelSettingSchema, + memoryModelsSchema, modelPreferHostedToolsConfigError, providerModelCostsConfigError, providerRelativeSendPathConfigError, @@ -160,6 +162,17 @@ export const configSchema = z.object({ modelPinnedEfforts: modelPinnedEffortsSchema.optional(), compactionRouting: compactionRoutingSchema.optional().catch(undefined), compactionRecovery: compactionRecoverySchema.optional().catch(undefined), + // A hand-edited malformed phase disables only that phase instead of rejecting + // providers/apiKeys, matching the load-time degradation notice; the management write + // boundary (validateConfigCandidate) still refuses the bad value through the shared, + // catch-free memoryModelsSchema. + memoryModels: z + .object({ + extract: memoryModelSettingSchema.optional().catch(undefined), + consolidation: memoryModelSettingSchema.optional().catch(undefined), + }) + .optional() + .catch(undefined), defaultProvider: z.string().min(1).default("openai"), defaultModelAliases: z.boolean().optional(), // Malformed hand edits disable this opt-in projection without rejecting providers. diff --git a/src/config/schema/leaf-validators.ts b/src/config/schema/leaf-validators.ts index 8de6060f5b2..a2d37384329 100644 --- a/src/config/schema/leaf-validators.ts +++ b/src/config/schema/leaf-validators.ts @@ -55,6 +55,21 @@ export const compactionRoutingSchema = z.object({ .optional(), }).strict(); +/** + * One phase of Codex's memory pipeline. A present phase must name a model: the GUI's "Off" + * removes the phase instead of blanking it, so an empty entry would only ever come from a + * hand-edited file, where failing the write is the honest answer. + */ +export const memoryModelSettingSchema = z.object({ + model: z.string().trim().min(1), + reasoningEffort: z.string().refine(value => pinnedReasoningEffortConfigError(value) === null).optional(), +}).strict(); + +export const memoryModelsSchema = z.object({ + extract: memoryModelSettingSchema.optional(), + consolidation: memoryModelSettingSchema.optional(), +}).strict(); + /** * Bounds for the opt-in same-target 429 wait-and-retry policy. Single source of truth * shared by the config schema, the load-time sanitizer, and the management write diff --git a/src/lib/shadow-call.ts b/src/lib/shadow-call.ts index 447cce52915..29117b48518 100644 --- a/src/lib/shadow-call.ts +++ b/src/lib/shadow-call.ts @@ -3,14 +3,17 @@ * * Codex 0.154.0+ sends `gpt-6-luna` for helper calls. Clients from 0.145.0 * through 0.153.x sent `gpt-5.6-luna`, which stays a default prefix so those - * clients keep their interception. Clients through 0.144.x used `gpt-5.4-mini`; - * operators supporting them can restore that prefix with the `sourceModels` - * override. The GPT-6 slug comes first because surfaces show the list in order. - * Every surface that names the + * clients keep their interception. `gpt-5.6-terra` is the model Codex asks for + * its background memory-consolidation pass, so an install that intercepts + * helper traffic keeps the whole memory pipeline off the native route instead + * of leaving that one phase on the account the operator routed away from. + * Clients through 0.144.x used `gpt-5.4-mini`; operators supporting them can + * restore that prefix with the `sourceModels` override. The order is the order + * the surfaces show. Every surface that names the * intercepted model (management API, GUI badges/tooltips, CLI) reads it from * here instead of hard-coding a slug that goes stale on the next client bump. */ -export const DEFAULT_SHADOW_SOURCE_MODELS = ["gpt-6-luna", "gpt-5.6-luna"] as const; +export const DEFAULT_SHADOW_SOURCE_MODELS = ["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"] as const; /** * Optional blocked model redirects at the shared routing layer. diff --git a/src/server/management/config-routes.ts b/src/server/management/config-routes.ts index 45cfee41f76..6b13547792a 100644 --- a/src/server/management/config-routes.ts +++ b/src/server/management/config-routes.ts @@ -1,4 +1,4 @@ -import { compactionRoutingSchema } from "../../config/schema/leaf-validators"; +import { compactionRoutingSchema, memoryModelsSchema } from "../../config/schema/leaf-validators"; import { compactionRecoverySchema } from "../../config/schema/compaction-recovery"; import { captureConfigTopLevelRollback } from "../../config/rebase-provenance"; import type { IntegrationClientId } from "../../integrations/registry"; @@ -374,6 +374,8 @@ export async function handleConfigRoutes(ctx: ManagementContext): Promise${applied.to}` : applied.to; + if (isInjectionDebugEnabled()) { + injectionDebugLog(`[opencodex] ${route.modelId}: memory ${phase} effort applied (${applied.from ?? "none"} -> ${applied.to})`); + } + } + } + } + { const { applyEffortCap, effortCapAppliesTo, supportedLadderFor } = await import("../effort-policy"); const surface = collabSurface(parsed); diff --git a/src/server/responses/core-options.ts b/src/server/responses/core-options.ts index a09bf122941..aea7f850800 100644 --- a/src/server/responses/core-options.ts +++ b/src/server/responses/core-options.ts @@ -145,6 +145,8 @@ export interface HandleResponsesOptions { comboAttempt?: boolean; /** Internal handoff: this combo was selected by shadow-call interception. */ shadowCallIntercepted?: boolean; + /** Internal handoff: the memory phase this turn belongs to, so combo children keep its routing. */ + memoryModelPhase?: "extract" | "consolidation"; compactionRoutingOverride?: CompactionRoutingOverride | null; /** Internal combo handoff for one parent-validated continuation snapshot. */ comboReplaySnapshot?: { diff --git a/src/server/responses/memory-models.ts b/src/server/responses/memory-models.ts new file mode 100644 index 00000000000..621875b6fd4 --- /dev/null +++ b/src/server/responses/memory-models.ts @@ -0,0 +1,164 @@ +/** + * Model routing for Codex's own memory pipeline. + * + * Codex writes memories in two background phases, and both ask the provider for a bare native + * model: Phase 1 ("extract") summarizes one finished thread per call and asks for + * `gpt-5.6-luna` at effort `low`; Phase 2 ("consolidation") is one agent run that merges those + * summaries into the files under `$CODEX_HOME/memories` and asks for `gpt-5.6-terra` at effort + * `medium`. Without a configured target both resolve through the canonical OpenAI route even when + * every ordinary turn is routed elsewhere — and Phase 1 additionally looks like the app's + * title/commit helper traffic, because the app uses the same model id for those. + * + * A phase is therefore recognized from Codex's own turn metadata, never inferred from the model + * id, the timing, or the token counts. Phase 1 sends `request_kind: "memory"`; both phases carry + * `thread_source: "memory_consolidation"`, and Phase 2 additionally arrives with + * `x-openai-subagent: memory_consolidation` (codex-rs `core/src/responses_metadata.rs`). + */ +import type { OcxConfig, OcxParsedRequest } from "../../types"; +import { isDeclaredReasoningEffort } from "../../reasoning-effort"; + +/** The two phases Codex runs, in the order it runs them. */ +export type MemoryModelPhase = "extract" | "consolidation"; + +/** codex-rs serializes both keys below into the JSON `x-codex-turn-metadata` header. */ +const TURN_METADATA_HEADER = "x-codex-turn-metadata"; +const REQUEST_KIND_KEY = "request_kind"; +const THREAD_SOURCE_KEY = "thread_source"; +/** `CodexResponsesRequestKind::Memory` (codex-rs `core/src/responses_metadata.rs`). */ +const MEMORY_REQUEST_KIND = "memory"; +/** `ThreadSource::MemoryConsolidation` / `InternalSessionSource::MemoryConsolidation`. */ +const MEMORY_THREAD_SOURCE = "memory_consolidation"; +const SUBAGENT_HEADER = "x-openai-subagent"; + +function record(value: unknown): Record | undefined { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? value as Record + : undefined; +} + +/** One metadata copy's verdict. `"none"` is a well-formed copy that is not a memory turn. */ +type CopyVerdict = MemoryModelPhase | "none"; + +function verdictOf(parsed: Record): CopyVerdict { + // Phase 1's detached request names the memory kind explicitly. Phase 2 is an ordinary turn + // inside the `memory_consolidation` thread, so its thread source is the only signal there. + if (parsed[REQUEST_KIND_KEY] === MEMORY_REQUEST_KIND) return "extract"; + if (parsed[THREAD_SOURCE_KEY] === MEMORY_THREAD_SOURCE) return "consolidation"; + return "none"; +} + +/** + * Recognize a memory-pipeline turn, or null. + * + * Every copy of the turn metadata the request carries must agree — the same rule + * `applyCompactionRoutingOverride` applies to compaction turns: a request that contradicts itself + * is not a memory turn, so neither copy can widen what the setting covers. On HTTP the sub-agent + * header is accepted on its own because Codex may deliver only that copy; on websocket it is not, + * because the bridge re-attaches the handshake's header to every frame, so there it marks the + * connection rather than the turn and the per-frame metadata decides alone. + */ +export function detectMemoryModelPhase( + body: unknown, + headers: Headers, + options: { transport?: "websocket" } = {}, +): MemoryModelPhase | null { + const metadata: unknown[] = []; + const header = headers.get(TURN_METADATA_HEADER); + if (options.transport !== "websocket" && header !== null) metadata.push(header); + const client = record(record(body)?.["client_metadata"]); + if (client && Object.hasOwn(client, TURN_METADATA_HEADER)) metadata.push(client[TURN_METADATA_HEADER]); + + let verdict: CopyVerdict | null = null; + for (const value of metadata) { + if (typeof value !== "string") return null; + let parsed: Record | undefined; + try { + parsed = record(JSON.parse(value)); + } catch { + return null; + } + if (!parsed) return null; + const copy = verdictOf(parsed); + if (verdict !== null && verdict !== copy) return null; + verdict = copy; + } + if (verdict === "extract" || verdict === "consolidation") return verdict; + // The websocket bridge rebuilds internal requests from a header allowlist and re-attaches the + // handshake's sub-agent header to every frame. Trusting it here would sweep the connection's + // later ordinary turns into the consolidation phase, so websocket frames rely on the per-frame + // turn metadata above and nothing else. + if (options.transport === "websocket") return null; + return headers.get(SUBAGENT_HEADER) === MEMORY_THREAD_SOURCE ? "consolidation" : null; +} + +/** The configured destination for one phase, or undefined while the phase keeps Codex's choice. */ +export function configuredMemoryModel( + config: Pick | undefined, + phase: MemoryModelPhase, +): { model: string; reasoningEffort?: string } | undefined { + const setting = config?.memoryModels?.[phase]; + if (!setting) return undefined; + const model = typeof setting.model === "string" ? setting.model.trim() : ""; + if (!model) return undefined; + const effort = typeof setting.reasoningEffort === "string" ? setting.reasoningEffort : undefined; + return { model, ...(effort ? { reasoningEffort: effort } : {}) }; +} + +/** + * Force the configured effort onto a memory turn. + * + * Codex hard-codes the phase effort (`low` for Phase 1, `medium` for Phase 2) and has no config + * key for it, so this is the only place the operator's choice can land. Both wire shapes are + * written: `parsed.options.reasoning` feeds the routed adapters, `_rawBody.reasoning.effort` feeds + * the ChatGPT passthrough serializer — the same dual-shape contract `applyPinnedEffort` uses. + */ +export function applyMemoryModelEffort( + parsed: OcxParsedRequest, + config: Pick | undefined, + phase: MemoryModelPhase, +): { from: string | undefined; to: string } | null { + const effort = configuredMemoryModel(config, phase)?.reasoningEffort; + if (!effort || !isDeclaredReasoningEffort(effort)) return null; + const requested = parsed.options.reasoning; + if (requested === effort) return null; + parsed.options.reasoning = effort; + const raw = parsed._rawBody as { reasoning?: { effort?: string } } | undefined; + if (raw && typeof raw === "object") { + raw.reasoning = { ...(record(raw.reasoning) ?? {}), effort } as { effort?: string }; + } + return { from: requested, to: effort }; +} + +/** Route reason recorded for a routed memory turn, so the request log names the phase. */ +export function memoryModelRouteReason(phase: MemoryModelPhase): string { + return phase === "extract" ? "memory-extract" : "memory-consolidation"; +} + +/** Non-retryable: the target stays unavailable until the operator changes the setting. */ +export const MEMORY_MODEL_TARGET_UNAVAILABLE_CODE = "memory_model_target_unavailable"; +export const MEMORY_MODEL_TARGET_UNAVAILABLE_STATUS = 409; + +const warnedTargets = new Set(); + +/** + * A configured phase destination that stopped resolving fails its call once, clearly, instead of + * silently falling back to the native model the operator routed away from — the same contract the + * shadow intercept uses for its single target. + */ +export function memoryModelTargetUnavailableResponse( + phase: MemoryModelPhase, + model: string, + detail: string, +): Response { + const message = `Memory ${phase} model "${model}" is unavailable: ${detail}. ` + + "Choose another model in the Memory Models settings or re-enable its provider."; + const key = `${phase}\u0000${model}\u0000${detail}`; + if (!warnedTargets.has(key)) { + warnedTargets.add(key); + console.warn(`memory-models: ${message}`); + } + return new Response( + JSON.stringify({ error: { message, type: "invalid_request_error", code: MEMORY_MODEL_TARGET_UNAVAILABLE_CODE } }), + { status: MEMORY_MODEL_TARGET_UNAVAILABLE_STATUS, headers: { "Content-Type": "application/json" } }, + ); +} diff --git a/src/server/responses/request-prepare.ts b/src/server/responses/request-prepare.ts index 2ac1679a72f..8d1c31b9197 100644 --- a/src/server/responses/request-prepare.ts +++ b/src/server/responses/request-prepare.ts @@ -13,7 +13,14 @@ import { } from "./core-errors"; import { parseSyntheticRowId } from "../fast-row"; import { resolveComboId, comboIdFromRawBody, NoAvailableComboTargetsError } from "../../combos"; -import { INTERCEPT_TARGET_UNAVAILABLE_CODE, interceptTargetUnavailableResponse, resolveShadowCallTarget } from "./shadow-target-availability"; +import { INTERCEPT_TARGET_UNAVAILABLE_CODE, interceptTargetUnavailableResponse, resolveChosenTarget, resolveShadowCallTarget } from "./shadow-target-availability"; +import { + MEMORY_MODEL_TARGET_UNAVAILABLE_CODE, + configuredMemoryModel, + detectMemoryModelPhase, + memoryModelRouteReason, + memoryModelTargetUnavailableResponse, +} from "./memory-models"; import { recallComboForLane } from "./combo-session-recall"; import { sessionLaneIdFromRequest, @@ -175,6 +182,19 @@ export async function prepareResponsesRequest( transport: options.inboundTransport, }); } + // Codex's memory pipeline names a destination per phase. The phase is read from Codex's own turn + // metadata, never from the model id: Phase 1 shares `gpt-5.6-luna` with the app's title/commit + // helper calls. Read here, ahead of the shadow intercept below, because the phase decision is the + // more specific of the two settings and must be the one that survives when both match one request. + const memoryModelPhase = options.memoryModelPhase + ?? (!options.comboAttempt && !options.compactionRoutingOverride && inboundWire === "responses" + ? detectMemoryModelPhase(body, req.headers, { transport: options.inboundTransport }) ?? undefined + : undefined); + const memoryModelTarget = memoryModelPhase ? configuredMemoryModel(config, memoryModelPhase) : undefined; + // A combo child is a synthetic replay of the parent's decision: its model is already the target's + // concrete provider/model, so neither site below may rewrite or re-resolve it. It keeps the phase + // through `options.memoryModelPhase` instead, which is what applies the phase effort. + const memoryModelApplies = memoryModelTarget !== undefined && options.comboAttempt !== true; options.onRequestBodyParsed?.(body); // An effort row naming a table-less combo (`combo/x--high`) must reach the combo dispatcher // as its base id, so the selector is normalized here, before comboIdFromRawBody reads model. @@ -229,11 +249,22 @@ export async function prepareResponsesRequest( // hops — which only exist inside that loop — are unreachable (#4129). Rewrite the selector // here instead, before comboIdFromRawBody reads `model`, and identify the combo by CONFIG // LOOKUP so the check can never observe a one-candidate collapse. + // A memory target that names a combo has to reach the combo dispatcher as `model`, or its own + // failover loop is unreachable (#4129) — the same reason the shadow intercept rewrites its combo + // target here. Every other target is resolved at the late site, where the admission scope exists. + let memoryModelComboRouted = false; + if (memoryModelApplies && memoryModelTarget && body && typeof body === "object" && !Array.isArray(body)) { + const memoryComboId = resolveComboId(config, memoryModelTarget.model); + if (memoryComboId && Object.hasOwn(config.combos ?? {}, memoryComboId)) { + memoryModelComboRouted = true; + (body as Record).model = memoryModelTarget.model; + } + } let shadowCallIntercepted = false; // A spawned sub-agent turn names its model on purpose; gpt-6-luna is both the helper // slug and a default sub-agent model, so neither intercept site may rewrite that turn. const threadSpawn = isThreadSpawnRequest(req.headers); - if (!options.comboAttempt && !options.compactionRoutingOverride && !threadSpawn && body && typeof body === "object" && !Array.isArray(body)) { + if (!options.comboAttempt && !options.compactionRoutingOverride && !threadSpawn && !memoryModelApplies && body && typeof body === "object" && !Array.isArray(body)) { const shadowIntercept = config.shadowCallIntercept; const rawShadowModel = (body as { model?: unknown }).model; if (shadowIntercept?.enabled && shadowIntercept.model && typeof rawShadowModel === "string" @@ -259,6 +290,9 @@ export async function prepareResponsesRequest( // Concrete combo child selectors no longer match the shadow source model. Carry the // interception decision explicitly so provider-specific helper isolation still applies. shadowCallIntercepted, + // Same handoff for a memory phase whose target is a combo: the child keeps the phase's effort + // override and stays out of the parent conversation. + memoryModelPhase: memoryModelComboRouted ? memoryModelPhase : undefined, // The original request body was accepted above. Combo children are synthetic // replays and must not repeat the caller-owned timeout transition. onRequestBodyRead: undefined, @@ -391,6 +425,10 @@ export async function prepareResponsesRequest( } if (cursorClientThreadId) parsed._cursorClientThreadId = cursorClientThreadId; if (options.shadowCallIntercepted === true) parsed._cursorIsolateConversation = true; + if (options.memoryModelPhase !== undefined) { + parsed._memoryModelPhase = options.memoryModelPhase; + parsed._cursorIsolateConversation = true; + } } catch (err) { if (isTranslatorBudgetExceededError(err)) { return formatErrorResponse(413, "request_too_large", "request translation buffer exceeded the safe limit", { @@ -494,9 +532,28 @@ export async function prepareResponsesRequest( : parsed._compactionRequest === true ? routeCompactionModel(config, modelId, evidenceFromBody(parsed._rawBody)) : routeModel(config, modelId, evidenceFromBody(parsed._rawBody))); + // The phase's destination. Resolved through the admission-scoped resolver every other route + // uses, and it fails closed exactly like the shadow target: falling back to the native model + // would spend the quota the operator routed away from, without their choosing it. + let memoryRoute: RouteResult | undefined; + if (memoryModelApplies && memoryModelPhase && memoryModelTarget) { + const memoryTarget = resolveChosenTarget(memoryModelTarget.model, resolveRoute); + if ("unavailable" in memoryTarget) { + logCtx.errorCode = MEMORY_MODEL_TARGET_UNAVAILABLE_CODE; + return memoryModelTargetUnavailableResponse(memoryModelPhase, memoryModelTarget.model, memoryTarget.unavailable); + } + credentialDomainWasRewritten = true; + parsed.modelId = memoryModelTarget.model; + if (parsed._rawBody && typeof parsed._rawBody === "object") { + (parsed._rawBody as { model?: string }).model = memoryModelTarget.model; + } + parsed._memoryModelPhase = memoryModelPhase; + parsed._cursorIsolateConversation = true; + memoryRoute = memoryTarget.route; + } const _sci = config.shadowCallIntercept; let shadowRoute: RouteResult | undefined; - if (!options.compactionRoutingOverride && !threadSpawn && _sci?.enabled && _sci.model && isShadowSourceModel(parsed.modelId, _sci.sourceModels)) { + if (!memoryRoute && !options.compactionRoutingOverride && !threadSpawn && _sci?.enabled && _sci.model && isShadowSourceModel(parsed.modelId, _sci.sourceModels)) { const sourcePrefix = shadowSourceModelPrefix(parsed.modelId, _sci.sourceModels)!; let sourceIdentity = { providerName: OPENAI_CODEX_PROVIDER_ID, modelId: sourcePrefix }; try { @@ -533,7 +590,11 @@ export async function prepareResponsesRequest( } } if (parsed._compactionRequest === true || options.compactionRoutingOverride) parsed._cursorIsolateConversation = true; - route = shadowRoute ?? resolveRoute(parsed.modelId); + route = memoryRoute ?? shadowRoute ?? resolveRoute(parsed.modelId); + // Name the phase in the persisted route decision, so the request log says why this turn went to + // the memory destination instead of leaving it looking like a plain user selection. Set here, on + // the resolved route, so a combo child's own route carries it too. + if (parsed._memoryModelPhase) route.routeReason = memoryModelRouteReason(parsed._memoryModelPhase); if (options.compactionRoutingOverride && !compactionRoutingKeepsProviderIdentity(config, options.compactionRoutingOverride, route)) { credentialDomainWasRewritten = true; // The destination does not share the conversation's credential domain, so it can neither diff --git a/src/server/responses/shadow-target-availability.ts b/src/server/responses/shadow-target-availability.ts index 814f99c9316..4c329fb2ff9 100644 --- a/src/server/responses/shadow-target-availability.ts +++ b/src/server/responses/shadow-target-availability.ts @@ -16,16 +16,21 @@ export const INTERCEPT_TARGET_UNAVAILABLE_CODE = "intercept_target_unavailable"; /** Non-retryable: the target stays unavailable until the operator changes the configuration. */ export const INTERCEPT_TARGET_UNAVAILABLE_STATUS = 409; -export type ShadowTargetResolution = { route: RouteResult } | { unavailable: string }; +export type ChosenTargetResolution = { route: RouteResult } | { unavailable: string }; +export type ShadowTargetResolution = ChosenTargetResolution; /** - * Resolve the configured target. Admission-scope refusals, exhausted combos and policy + * Resolve one operator-chosen target. Admission-scope refusals, exhausted combos and policy * evaluations keep their existing responses, so they are rethrown to the caller's handler. + * + * Shared by the shadow-call intercept and the memory-model routing: both name a single destination + * whose unavailability must not be papered over by the router's terminal default-provider + * fallback. `shadowCallTargetsIntersect` and the memory setting are the two callers. */ -export function resolveShadowCallTarget( +export function resolveChosenTarget( model: string, resolve: (model: string) => RouteResult, -): ShadowTargetResolution { +): ChosenTargetResolution { let route: RouteResult; try { route = resolve(model); @@ -44,6 +49,14 @@ export function resolveShadowCallTarget( return { route }; } +/** The shadow-call name for the shared resolver, kept because that is the surface's own vocabulary. */ +export function resolveShadowCallTarget( + model: string, + resolve: (model: string) => RouteResult, +): ShadowTargetResolution { + return resolveChosenTarget(model, resolve); +} + const warnedTargets = new Set(); export function interceptTargetUnavailableResponse(model: string, detail: string): Response { diff --git a/src/types/config.ts b/src/types/config.ts index 940aa2b06fd..fafa2ca38ab 100644 --- a/src/types/config.ts +++ b/src/types/config.ts @@ -738,6 +738,27 @@ export interface OcxConfig { }; /** Opt-in failure-only recovery; never replaces the initial compaction model. */ compactionRecovery?: { enabled: boolean; model: string; allowDevinInvalidArgument?: boolean }; + /** + * Destination model for Codex's own memory pipeline, per phase + * (src/server/responses/memory-models.ts). + * + * Codex runs Phase 1 ("extract") once per finished thread to summarize that thread's rollout, + * and Phase 2 ("consolidation") once as an agent run that merges the summaries into the files + * under `$CODEX_HOME/memories`. Both ask for a bare native model, so without an entry here they + * resolve through the canonical OpenAI route even when ordinary turns are routed elsewhere. + * + * A phase is recognized from Codex's turn metadata, never inferred from the model id, the timing + * or the token counts: Phase 1 shares `gpt-5.6-luna` with the app's title/commit helper calls, + * and `shadowCallIntercept` is the setting for those. A configured phase wins over that + * intercept, because the memory decision is the more specific one. + * + * `model` is required for a configured phase; omitting the phase (or its `model`) leaves Codex's + * own choice in place. `reasoningEffort` overrides the effort Codex hard-codes for that phase. + */ + memoryModels?: { + extract?: { model: string; reasoningEffort?: string }; + consolidation?: { model: string; reasoningEffort?: string }; + }; /** * Models hidden from Codex discovery without blocking direct proxy calls. Routed provider ids * are excluded from the catalog + /v1/models entirely. Account-qualified native ids hide only @@ -756,7 +777,8 @@ export interface OcxConfig { /** * Shadow call intercept: redirect Codex's hard-coded helper calls (title generation, * commit messages, skill orchestration) to a user-chosen model. Default intercepted - * source models: gpt-6-luna (Codex 0.154.0+) and gpt-5.6-luna (0.145.0-0.153.x). + * source models: gpt-6-luna (Codex 0.154.0+), gpt-5.6-luna (0.145.0-0.153.x), and + * gpt-5.6-terra, which Codex asks for its background memory-consolidation pass. * Clients through 0.144.x emitted gpt-5.4-mini instead; that model is retired upstream, * but it stays available as an opt-in `sourceModels` prefix so an old client's helper * calls can still be intercepted. @@ -771,7 +793,7 @@ export interface OcxConfig { enabled?: boolean; /** Replacement model id (e.g. "gpt-5.5"). */ model?: string; - /** Optional override of intercepted source-model prefixes (default: gpt-6-luna, gpt-5.6-luna). */ + /** Optional override of intercepted source-model prefixes (default: gpt-6-luna, gpt-5.6-luna, gpt-5.6-terra). */ sourceModels?: string[]; }; /** diff --git a/src/types/request.ts b/src/types/request.ts index 03faeb5afec..288de9383f4 100644 --- a/src/types/request.ts +++ b/src/types/request.ts @@ -143,6 +143,12 @@ export interface OcxParsedRequest { _compactionRequest?: boolean; /** Manual compaction moved to another provider: summarize portably even on a canonical ChatGPT target. */ _portableCompaction?: boolean; + /** + * Codex memory pipeline phase this turn belongs to, when `memoryModels` routes it + * (src/server/responses/memory-models.ts). Read at the effort choke point, which runs after the + * route is known. + */ + _memoryModelPhase?: "extract" | "consolidation"; /** * True when the current request newly introduced a stored compaction summary/marker. Historical * markers restored by previous_response_id expansion were already acknowledged and do not reset diff --git a/structure/gui-and-management-api.md b/structure/gui-and-management-api.md index 56ea1763b01..55bde38ce3e 100644 --- a/structure/gui-and-management-api.md +++ b/structure/gui-and-management-api.md @@ -233,7 +233,7 @@ per-request first-party callback reads that live object; a failed write leaves i | System | `POST /api/system/restart` restarts the proxy in place. Local CLI/tray callers first attest the exact runtime PID and port, then send a process-scoped HMAC capability bound to that method, path, PID, and port; the capability authorizes no other management route and is invalid after replacement. The caller observes one absolute deadline and accepts success only after a different runtime PID is healthy on the same port. `GET /api/system/health` is the authenticated scalar-only identity used by shared-plane Dashboard status and restart reconnect polling; its `spendLedger` block reports only ownership held/unheld, initialized/configured/degraded booleans and bounded persistence/corruption counters. Reading it never constructs, replays or prunes the ledger. Paths, scopes, accounts and request ids are absent, and the block never moves to unauthenticated `/healthz`. `GET /api/system/memory` — service-process runtime/memory identity (pid, Bun version/revision, optional `bunRuntimeSource` provenance, platform, RSS/heap/external/ArrayBuffers scalars, observed memory = max(RSS, external, ArrayBuffers), `bun:jsc` heap context, streamMode + eager-relay gate decision, watchdog snapshot sliced to the last 60 samples) plus privacy-safe `appOwnedBytes` retained-store totals/counters under static store ids. Its response-state block also reports spill-write `initial`/`healthy`/`degraded` status, a consecutive-failure streak, fixed error class, and failure/success timestamps. A successful publication clears the streak in the same process; raw error text and paths never enter this surface. Scalar-only payload; dashboard/admin callers use the standard management gate, while `ocx doctor` may use only the exact process-scoped local-read capability. It must never move to unauthenticated `/healthz`. | | Stop | `POST /api/stop` — restore native Codex, stop any installed service, and exit the proxy. A sibling instance restores nothing and answers `sharedTeardown: "not-owned"` ([Codex home](codex-home.md#codex-home)). | | Diagnostics/sync | `src/server/management/config-routes.ts` — `GET /api/diagnostics/project-config` reports project-level Codex config that bypasses managed routing; `POST /api/sync` re-runs catalog/config sync. The diagnostic reports the bypass; it does not rewrite the project file. | -| Sidecar/shadow-call settings | `src/server/management/config-routes.ts` — `GET/PUT /api/sidecar-settings` and `GET/PUT /api/shadow-call-settings`. PUT accepts model and backend (web-search union: openai/anthropic/xai/gemini/exa; xAI is live through stored Grok OAuth, while Gemini/Exa remain inert until their executors ship) plus validated `webSearch.xSearch`, optional `webSearch.exaApiKey` (write/clear only — never echoed by GET or the PUT response; redact.ts strips it from logs), `webSearch.reasoning`, `vision.reasoning`, `vision.enabled`, `vision.maxDescriptionsPerTurn`, and `vision.timeoutMs`; the read and PUT-response payload reports model, backend, reasoning, enabled, the vision per-turn limit, and timeout. `timeoutMs` is validated against the runtime integer bounds in `src/vision/timeout-bounds.ts`. Provider/OAuth credentials live in their stores; `exaApiKey` is the one sidecar-owned secret and follows the write-only contract above. Both shadow-call responses also report the resolved `sourceModels` — the prefixes the runtime actually intercepts (`src/lib/shadow-call.ts`, default `gpt-6-luna` and `gpt-5.6-luna`; the retired `gpt-5.4-mini` stays available as an explicit `sourceModels` entry for 0.144.x clients), so no client hard-codes a helper slug that a Codex release can invalidate. PUT refuses a qualified target that only the router's default-provider fallback accepts. Provider disable and delete report the target they leave behind as `dependentShadowIntercept` (`shadowInterceptProviderDependency` in `src/server/management/shadow-call-validation.ts`), and at request time an unresolvable target returns `409 intercept_target_unavailable` before any send (`src/server/responses/shadow-target-availability.ts`); it never falls back to the native source model or the default provider. | +| Sidecar/shadow-call settings | `src/server/management/config-routes.ts` — `GET/PUT /api/sidecar-settings` and `GET/PUT /api/shadow-call-settings`. PUT accepts model and backend (web-search union: openai/anthropic/xai/gemini/exa; xAI is live through stored Grok OAuth, while Gemini/Exa remain inert until their executors ship) plus validated `webSearch.xSearch`, optional `webSearch.exaApiKey` (write/clear only — never echoed by GET or the PUT response; redact.ts strips it from logs), `webSearch.reasoning`, `vision.reasoning`, `vision.enabled`, `vision.maxDescriptionsPerTurn`, and `vision.timeoutMs`; the read and PUT-response payload reports model, backend, reasoning, enabled, the vision per-turn limit, and timeout. `timeoutMs` is validated against the runtime integer bounds in `src/vision/timeout-bounds.ts`. Provider/OAuth credentials live in their stores; `exaApiKey` is the one sidecar-owned secret and follows the write-only contract above. Both shadow-call responses also report the resolved `sourceModels` — the prefixes the runtime actually intercepts (`src/lib/shadow-call.ts`, default `gpt-6-luna`, `gpt-5.6-luna` and `gpt-5.6-terra` (Codex's background memory-consolidation model); the retired `gpt-5.4-mini` stays available as an explicit `sourceModels` entry for 0.144.x clients), so no client hard-codes a helper slug that a Codex release can invalidate. PUT refuses a qualified target that only the router's default-provider fallback accepts. Provider disable and delete report the target they leave behind as `dependentShadowIntercept` (`shadowInterceptProviderDependency` in `src/server/management/shadow-call-validation.ts`), and at request time an unresolvable target returns `409 intercept_target_unavailable` before any send (`src/server/responses/shadow-target-availability.ts`); it never falls back to the native source model or the default provider. | | Storage | `src/server/management/logs-usage-routes.ts` — `GET /api/storage`, `POST /api/storage/cleanup/preview` and `/api/storage/cleanup`, `GET /api/storage/trash`, `POST /api/storage/trash/restore`, and `GET/PUT /api/storage/cleanup-policy` plus `POST /api/storage/cleanup-policy/run`. `GET /api/storage/cleanup-policy/test-stream` and `GET /api/storage/trash/restore/test-stream` exist for progress-stream testing. Cleanup takes an explicit `mode`: `quarantine` moves to trash and is restorable, `permanent` is not. The caller must name the mode — there is no default that silently deletes. | | Provider quotas and tests | `src/server/management/provider-routes.ts` — `GET /api/provider-quotas`, `POST /api/providers/test`, `GET/PUT /api/provider-context-caps`, `GET /api/provider-presets`. A quota read may be served from cache or force-refreshed; absent quota data is reported as unknown rather than as a measured zero. | | Models and visibility | `src/server/management/model-routes.ts` — `GET /api/models`, `PUT /api/disabled-models`, `PUT /api/model-visibility`, `PUT /api/selected-models`, `GET/POST /api/custom-models`. Visibility writes trigger catalog sync through the owning server path. | diff --git a/tests/config/settings-memory-models.test.ts b/tests/config/settings-memory-models.test.ts new file mode 100644 index 00000000000..b2648921738 --- /dev/null +++ b/tests/config/settings-memory-models.test.ts @@ -0,0 +1,99 @@ +/** + * `/api/settings` round-trip for per-phase memory routing. + * + * GET reports the block, PUT persists it to config.json and echoes it in its own response, + * and a fresh `loadConfig()` reads it back. The echo is load-bearing: the dashboard panel + * re-reads the response of its own save, so a response without the block would render both + * phases as "Off" while the server still held them. + */ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { mkdtempSync, readFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { getConfigPath, loadConfig, saveConfig } from "../../src/config"; +import { handleManagementAPI, type ManagementApiDeps } from "../../src/server/management-api"; +import { invalidateStartupHealthCache } from "../../src/server/startup-health-cache"; +import type { OcxConfig } from "../../src/types"; +import { startupHealthFixture } from "../helpers/startup-health"; +import { removeTreeWithRetry } from "../helpers/remove-tree"; + +let home = ""; +let previousHome: string | undefined; + +const baseConfig = (): OcxConfig => ({ + port: 10100, + defaultProvider: "gateway", + providers: { gateway: { adapter: "openai-chat", baseUrl: "https://gateway.test/v1", apiKey: "fixture" } }, +}); + +function settings(cfg: OcxConfig, body?: unknown) { + const req = new Request("http://127.0.0.1:10100/api/settings", { + method: body === undefined ? "GET" : "PUT", + headers: { host: "127.0.0.1:10100", "content-type": "application/json" }, + ...(body === undefined ? {} : { body: JSON.stringify(body) }), + }); + const deps: Partial = { getCachedStartupHealth: async () => startupHealthFixture() }; + return handleManagementAPI(req, new URL(req.url), cfg, deps); +} + +beforeEach(() => { + previousHome = process.env.OPENCODEX_HOME; + home = mkdtempSync(join(tmpdir(), "ocx-memory-models-settings-")); + process.env.OPENCODEX_HOME = home; + invalidateStartupHealthCache(); +}); + +afterEach(() => { + invalidateStartupHealthCache(); + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + removeTreeWithRetry(home); +}); + +describe("/api/settings memoryModels", () => { + test("an unconfigured install reports no memory routing", async () => { + const response = await settings(baseConfig()); + expect(await response!.json()).toMatchObject({ memoryModels: null }); + }); + + test("a save is echoed in the PUT response, persisted, and reported by GET", async () => { + const config = baseConfig(); + saveConfig(config); + const setting = { + extract: { model: "gateway/cheap" }, + consolidation: { model: "gateway/strong", reasoningEffort: "high" }, + }; + const put = await settings(config, { memoryModels: setting }); + expect(put!.status).toBe(200); + expect(await put!.json()).toMatchObject({ ok: true, memoryModels: setting }); + expect(config.memoryModels).toEqual(setting); + expect(loadConfig().memoryModels).toEqual(setting); + expect(await (await settings(config))!.json()).toMatchObject({ memoryModels: setting }); + }); + + test("null clears the block from the file and from the response", async () => { + const config = baseConfig(); + config.memoryModels = { extract: { model: "gateway/cheap" } }; + saveConfig(config); + const put = await settings(config, { memoryModels: null }); + expect(put!.status).toBe(200); + expect(await put!.json()).toMatchObject({ memoryModels: null }); + expect(config.memoryModels).toBeUndefined(); + expect(loadConfig().memoryModels).toBeUndefined(); + const raw = JSON.parse(readFileSync(getConfigPath(), "utf8")) as Record; + expect(Object.hasOwn(raw, "memoryModels")).toBe(false); + }); + + test("a malformed phase is rejected before any mutation", async () => { + const config = baseConfig(); + config.memoryModels = { extract: { model: "gateway/cheap" } }; + saveConfig(config); + const before = structuredClone(config); + for (const value of [{ extract: { model: " " } }, { extract: { model: "m", reasoningEffort: "bogus" } }, + { extract: { model: "m", extra: true } }, { extract: "gateway/cheap" }]) { + const response = await settings(config, { memoryModels: value }); + expect(response!.status).toBe(400); + expect(config).toEqual(before); + } + }); +}); diff --git a/tests/helpers/responses-core-source.ts b/tests/helpers/responses-core-source.ts index f803be42c3e..6d40d36e466 100644 --- a/tests/helpers/responses-core-source.ts +++ b/tests/helpers/responses-core-source.ts @@ -35,6 +35,7 @@ export const RESPONSES_CORE_MODULES = [ "compaction-routing.ts", "compaction-recovery.ts", "compaction-recovery-policy.ts", + "memory-models.ts", "request-transport.ts", "request-sidecar-auth.ts", "response-effects.ts", diff --git a/tests/responses/responses-memory-models.test.ts b/tests/responses/responses-memory-models.test.ts new file mode 100644 index 00000000000..363afe45ea5 --- /dev/null +++ b/tests/responses/responses-memory-models.test.ts @@ -0,0 +1,373 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { + MEMORY_MODEL_TARGET_UNAVAILABLE_CODE, + applyMemoryModelEffort, + configuredMemoryModel, + detectMemoryModelPhase, +} from "../../src/server/responses/memory-models"; +import { handleResponses } from "../../src/server/responses"; +import { MODEL_NOT_ALLOWED_FOR_KEY } from "../../src/server/admission-model-scope"; +import { getDefaultConfig, validateConfigCandidate } from "../../src/config"; +import { configSchema } from "../../src/config/schema/config-schema"; +import { warnDegradedMemoryModels } from "../../src/config/load-degrade"; +import { clearComboSelectionState, clearComboTargetCooldowns } from "../../src/combos"; +import { acquireOwnedSpendHome } from "../helpers/owned-spend-home"; +import type { OcxConfig, OcxParsedRequest } from "../../src/types"; + +const originalFetch = globalThis.fetch; +/** The spend-journal writer lease is taken by startServer, so a bare handler call needs one. */ +let releaseSpendHome: (() => void) | undefined; + +/** Phase 1's shape: codex-rs marks the kind AND the thread source. */ +const extractMetadata = (extra: Record = {}) => + JSON.stringify({ request_kind: "memory", thread_source: "memory_consolidation", ...extra }); +/** Phase 2's shape: an ordinary turn inside the consolidation thread. */ +const consolidationMetadata = () => + JSON.stringify({ request_kind: "turn", thread_source: "memory_consolidation" }); + +function config(): OcxConfig { + return { + ...getDefaultConfig(), + defaultProvider: "gateway", + providers: { + gateway: { + adapter: "openai-responses", authMode: "key", + baseUrl: "https://gateway.example/v1", apiKey: "fixture-key", + }, + }, + memoryModels: { + extract: { model: "gateway/cheap", reasoningEffort: "high" }, + consolidation: { model: "gateway/strong", reasoningEffort: "xhigh" }, + }, + }; +} + +function body(model = "gpt-5.6-luna"): Record { + return { + model, stream: false, + reasoning: { effort: "low", summary: "auto" }, + input: [{ type: "message", role: "user", content: [{ type: "input_text", text: "Summarize this rollout." }] }], + }; +} + +function request(value: unknown, metadata?: string, extraHeaders: Record = {}): Request { + return new Request("http://localhost/v1/responses", { + method: "POST", + headers: { + "content-type": "application/json", session_id: "memory-models-fixture", + ...(metadata ? { "x-codex-turn-metadata": metadata } : {}), + ...extraHeaders, + }, + body: JSON.stringify(value), + }); +} + +function completion(): Record { + return { + id: "resp_memory_fixture", status: "completed", + output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "ok" }] }], + usage: { input_tokens: 10, output_tokens: 5, total_tokens: 15 }, + }; +} + +beforeEach(() => { + releaseSpendHome = acquireOwnedSpendHome(); +}); + +afterEach(() => { + releaseSpendHome?.(); + releaseSpendHome = undefined; + globalThis.fetch = originalFetch; + clearComboSelectionState(); + clearComboTargetCooldowns(); +}); + +describe("memory phase detection", () => { + test("recognizes each phase from Codex's own turn metadata", () => { + expect(detectMemoryModelPhase(body(), new Headers({ "x-codex-turn-metadata": extractMetadata() }))).toBe("extract"); + expect(detectMemoryModelPhase(body("gpt-5.6-terra"), new Headers({ "x-codex-turn-metadata": consolidationMetadata() }))).toBe("consolidation"); + }); + + test("takes the phase from the sub-agent header when the metadata copy carries none", () => { + const headers = new Headers({ + "x-codex-turn-metadata": JSON.stringify({ request_kind: "turn" }), + "x-openai-subagent": "memory_consolidation", + }); + expect(detectMemoryModelPhase(body("gpt-5.6-terra"), headers)).toBe("consolidation"); + // Any other internal turn category is not a memory turn. + expect(detectMemoryModelPhase(body(), new Headers({ "x-openai-subagent": "collab_spawn" }))).toBeNull(); + expect(detectMemoryModelPhase(body(), new Headers({ "x-openai-subagent": "review" }))).toBeNull(); + }); + + test("an ordinary turn, absent metadata, or malformed metadata is never a memory turn", () => { + expect(detectMemoryModelPhase(body(), new Headers())).toBeNull(); + expect(detectMemoryModelPhase(body(), new Headers({ "x-codex-turn-metadata": JSON.stringify({ request_kind: "turn", thread_source: "cli" }) }))).toBeNull(); + for (const value of ["{", "null", "[]", '"memory"', JSON.stringify({ request_kind: "memory_consolidation" })]) { + expect(detectMemoryModelPhase(body(), new Headers({ "x-codex-turn-metadata": value }))).toBeNull(); + } + // A non-string copy is malformed rather than absent. + expect(detectMemoryModelPhase({ ...body(), client_metadata: { "x-codex-turn-metadata": 42 } }, new Headers())).toBeNull(); + }); + + test("conflicting copies are not treated as a memory turn", () => { + for (const [header, embedded] of [[extractMetadata(), consolidationMetadata()], [consolidationMetadata(), extractMetadata()], [extractMetadata(), "{"], ["{", extractMetadata()]]) { + const input = { ...body(), client_metadata: { "x-codex-turn-metadata": embedded } }; + expect(detectMemoryModelPhase(input, new Headers({ "x-codex-turn-metadata": header! }))).toBeNull(); + } + }); + + test("both copies must agree on the same phase", () => { + const input = { ...body(), client_metadata: { "x-codex-turn-metadata": extractMetadata() } }; + expect(detectMemoryModelPhase(input, new Headers({ "x-codex-turn-metadata": extractMetadata() }))).toBe("extract"); + }); + + test("WebSocket frames read the body copy instead of the handshake header", () => { + const input = { ...body("gpt-5.6-terra"), client_metadata: { "x-codex-turn-metadata": consolidationMetadata() } }; + const headers = new Headers({ "x-codex-turn-metadata": extractMetadata() }); + expect(detectMemoryModelPhase(input, headers)).toBeNull(); + expect(detectMemoryModelPhase(input, headers, { transport: "websocket" })).toBe("consolidation"); + }); + + test("the connection's sub-agent header consolidates HTTP turns but not websocket frames", () => { + const headers = new Headers({ "x-openai-subagent": "memory_consolidation" }); + expect(detectMemoryModelPhase(body("gpt-5.6-terra"), headers)).toBe("consolidation"); + expect(detectMemoryModelPhase(body("gpt-5.6-terra"), headers, { transport: "websocket" })).toBeNull(); + }); +}); + +describe("memory model settings", () => { + test("a phase without a model is off, and a blank model is not a destination", () => { + const settings = config(); + expect(configuredMemoryModel(settings, "extract")).toEqual({ model: "gateway/cheap", reasoningEffort: "high" }); + delete settings.memoryModels!.consolidation; + expect(configuredMemoryModel(settings, "consolidation")).toBeUndefined(); + settings.memoryModels = { extract: { model: " " } }; + expect(configuredMemoryModel(settings, "extract")).toBeUndefined(); + expect(configuredMemoryModel(undefined, "extract")).toBeUndefined(); + }); + + test("the configured effort is written to both wire shapes", () => { + const parsed = { modelId: "gpt-5.6-luna", options: { reasoning: "low" }, _rawBody: { reasoning: { effort: "low", summary: "auto" } } } as unknown as OcxParsedRequest; + expect(applyMemoryModelEffort(parsed, config(), "extract")).toEqual({ from: "low", to: "high" }); + expect(parsed.options.reasoning).toBe("high"); + expect(parsed._rawBody!.reasoning).toEqual({ effort: "high", summary: "auto" }); + // Idempotent, and a phase without an effort leaves Codex's own value alone. + expect(applyMemoryModelEffort(parsed, config(), "extract")).toBeNull(); + expect(applyMemoryModelEffort(parsed, config(), "consolidation")).toEqual({ from: "high", to: "xhigh" }); + const bare = config(); + bare.memoryModels = { extract: { model: "gateway/cheap" } }; + const untouched = { modelId: "gpt-5.6-luna", options: { reasoning: "low" }, _rawBody: {} } as unknown as OcxParsedRequest; + expect(applyMemoryModelEffort(untouched, bare, "extract")).toBeNull(); + expect(untouched.options.reasoning).toBe("low"); + }); +}); + +describe("memory model config", () => { + test("validates both phases without resetting providers on malformed hand edits", () => { + expect(validateConfigCandidate(config()).ok).toBe(true); + for (const value of [null, [], "cheap", { extract: { model: " " } }, { extract: { model: 42 } }, + { consolidation: { model: "gateway/strong", reasoningEffort: "fast" } }, + { extract: { model: "gateway/cheap", typo: true } }, { extract: {}, unknown: true }]) { + const raw = { ...config(), memoryModels: value }; + expect(validateConfigCandidate(raw).ok).toBe(false); + const loaded = configSchema.parse(raw); + expect(loaded.providers).toEqual(config().providers); + } + // A wholly broken block drops entirely; a broken phase drops only that phase, so a typo in + // one phase can no longer delete the operator's routing for the other. + for (const value of [null, [], "cheap"]) { + const loaded = configSchema.parse({ ...config(), memoryModels: value }); + expect(loaded.memoryModels).toBeUndefined(); + } + const oneBroken = configSchema.parse({ ...config(), memoryModels: { extract: { model: " " }, consolidation: { model: "gateway/strong" } } }); + expect(oneBroken.memoryModels).toEqual({ consolidation: { model: "gateway/strong" } }); + }); + + test("an empty block is valid and means both phases stay with Codex", () => { + const raw = { ...config(), memoryModels: {} }; + expect(validateConfigCandidate(raw).ok).toBe(true); + expect(configuredMemoryModel(configSchema.parse(raw) as OcxConfig, "extract")).toBeUndefined(); + }); + + test("a broken phase warns per phase at load; valid or absent blocks stay silent", () => { + const warnings: string[] = []; + const original = console.warn; + console.warn = (message: unknown) => { warnings.push(String(message)); }; + try { + const invalid = { ...config(), memoryModels: { extract: { model: "" } } }; + warnDegradedMemoryModels(invalid, configSchema.parse(invalid) as OcxConfig); + expect(warnings).toHaveLength(1); + expect(warnings[0]).toContain("memoryModels.extract is invalid"); + // The route a broken phase keeps is not necessarily Codex's own model: the shadow-call + // intercept can still match the request. + expect(warnings[0]).toContain("keeps its existing route"); + // The surviving phase is not repeated, and a wholly broken block keeps the block wording. + const survivor = { ...config(), memoryModels: { extract: { model: "" }, consolidation: { model: "gateway/strong" } } }; + warnDegradedMemoryModels(survivor, configSchema.parse(survivor) as OcxConfig); + expect(warnings).toHaveLength(2); + expect(warnings[1]).toContain("memoryModels.extract is invalid"); + const whole = { ...config(), memoryModels: "cheap" }; + warnDegradedMemoryModels(whole, configSchema.parse(whole) as OcxConfig); + expect(warnings).toHaveLength(3); + expect(warnings[2]).toContain("memoryModels is invalid"); + warnDegradedMemoryModels(config(), configSchema.parse(config()) as OcxConfig); + const absent = config(); + delete absent.memoryModels; + warnDegradedMemoryModels(absent, configSchema.parse(absent) as OcxConfig); + expect(warnings).toHaveLength(3); + } finally { + console.warn = original; + } + }); + + test("an unrecognized phase key warns instead of vanishing on the next save", () => { + const warnings: string[] = []; + const original = console.warn; + console.warn = (message: unknown) => { warnings.push(String(message)); }; + try { + const typo = { ...config(), memoryModels: { extrcat: { model: "gateway/cheap" }, consolidation: { model: "gateway/strong" } } }; + const parsed = configSchema.parse(typo) as OcxConfig; + // The load schema stays permissive, so the misspelled key is stripped while the valid phase + // survives - which is exactly why the warning has to read the raw object. + expect(parsed.memoryModels).toEqual({ consolidation: { model: "gateway/strong" } }); + warnDegradedMemoryModels(typo, parsed); + expect(warnings).toHaveLength(1); + // The key name is JSON-quoted because it is redacted and escaped before it reaches the log. + expect(warnings[0]).toContain('memoryModels."extrcat" is not a recognized phase'); + } finally { + console.warn = original; + } + }); +}); + +describe("memory model routing", () => { + test("routes each phase to its own model and effort", async () => { + const settings = config(); + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + + const extractCtx = { model: "", provider: "" } as { model: string; provider: string; requestedModel?: string }; + const extract = await handleResponses(request(body(), extractMetadata()), settings, extractCtx); + expect(extract.status).toBe(200); + await extract.text(); + expect(calls[0]!.model).toBe("cheap"); + expect(calls[0]!.reasoning.effort).toBe("high"); + // The caller's own selector stays in the log; only the served model changed. + expect(extractCtx.requestedModel).toBe("gpt-5.6-luna"); + expect(extractCtx.model).toBe("cheap"); + + const consolidationCtx = { model: "", provider: "" } as { model: string; provider: string; requestedModel?: string }; + const consolidation = await handleResponses(request(body("gpt-5.6-terra"), consolidationMetadata()), settings, consolidationCtx); + expect(consolidation.status).toBe(200); + await consolidation.text(); + expect(calls[1]!.model).toBe("strong"); + expect(calls[1]!.reasoning.effort).toBe("xhigh"); + expect(consolidationCtx.requestedModel).toBe("gpt-5.6-terra"); + }); + + test("an unconfigured phase and a turn without the marker keep their own model", async () => { + const settings = config(); + delete settings.memoryModels!.consolidation; + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + + // Phase 2 with only Phase 1 configured, then a Phase 1 turn with nothing configured. + const consolidation = await handleResponses(request(body("gateway/normal"), consolidationMetadata()), settings, { model: "", provider: "" }); + expect(consolidation.status).toBe(200); + await consolidation.text(); + const configured = config(); + delete configured.memoryModels; + const unconfigured = await handleResponses(request(body("gateway/normal"), extractMetadata()), configured, { model: "", provider: "" }); + expect(unconfigured.status).toBe(200); + await unconfigured.text(); + // Same model id, no marker: nothing about the phase may reach it. + const ordinary = await handleResponses(request(body("gateway/normal")), settings, { model: "", provider: "" }); + expect(ordinary.status).toBe(200); + await ordinary.text(); + expect(calls.map(call => [call.model, call.reasoning.effort])).toEqual([ + ["normal", "low"], ["normal", "low"], ["normal", "low"], + ]); + }); + + test("a memory turn keeps the phase decision when the shadow intercept would match too", async () => { + const settings = config(); + settings.shadowCallIntercept = { enabled: true, model: "gateway/helper" }; + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + const logCtx = { model: "", provider: "" } as { model: string; provider: string; shadowCallRewrittenFrom?: string }; + const response = await handleResponses(request(body(), extractMetadata()), settings, logCtx); + expect(response.status).toBe(200); + await response.text(); + expect(calls[0]!.model).toBe("cheap"); + expect(calls[0]!.reasoning.effort).toBe("high"); + // Phase 1 shares its model id with the app's helper calls, so the marker is what tells them apart. + expect(logCtx.shadowCallRewrittenFrom).toBeUndefined(); + }); + + test("a target that no longer resolves fails the memory call instead of falling back", async () => { + const settings = config(); + settings.memoryModels = { extract: { model: "ghost/cheap" } }; + const calls: string[] = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body)).model); + return Response.json(completion()); + }) as typeof fetch; + const response = await handleResponses(request(body(), extractMetadata()), settings, { model: "", provider: "" }); + expect(response.status).toBe(409); + expect((await response.json() as { error: { code: string } }).error.code).toBe(MEMORY_MODEL_TARGET_UNAVAILABLE_CODE); + expect(calls).toEqual([]); + }); + + test("an admission denial on the memory target keeps the key's own refusal", async () => { + const settings = config(); + settings.apiKeys = [{ + id: "scoped", name: "mail", key: "ocx_data_" + "c".repeat(40), + createdAt: "2026-01-01T00:00:00.000Z", allowedProviders: ["elsewhere"], + }]; + const calls: string[] = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body)).model); + return Response.json(completion()); + }) as typeof fetch; + const response = await handleResponses( + request(body(), extractMetadata()), + settings, + { model: "", provider: "" }, + { admission: { kind: "configured", keyId: "scoped", source: "bearer" } }, + ); + // The shared resolver rethrows an admission refusal, so the memory path reports the key's + // scope instead of turning it into an unavailable target. + expect(response.status).toBe(403); + expect((await response.json() as { error: { type: string } }).error.type).toBe(MODEL_NOT_ALLOWED_FOR_KEY); + expect(calls).toEqual([]); + }); + + test("the phase decision survives the combo handoff", async () => { + const settings = config(); + settings.combos = { memory: { targets: [{ provider: "gateway", model: "cheap" }] } }; + settings.memoryModels = { extract: { model: "combo/memory", reasoningEffort: "high" } }; + const calls: Array> = []; + globalThis.fetch = (async (_input: unknown, init?: RequestInit) => { + calls.push(JSON.parse(String(init?.body))); + return Response.json(completion()); + }) as typeof fetch; + const logCtx = { model: "", provider: "" } as { model: string; provider: string; requestedModel?: string }; + const response = await handleResponses(request(body(), extractMetadata()), settings, logCtx); + expect(response.status).toBe(200); + await response.text(); + expect(calls[0]!.model).toBe("cheap"); + expect(calls[0]!.reasoning.effort).toBe("high"); + // A combo target has to reach the dispatcher as `model`, so the rewritten selector is what the + // log records as requested; the phase itself is named in the route decision. + expect(logCtx.requestedModel).toBe("combo/memory"); + }); +}); diff --git a/tests/responses/responses-shadow-intercept.test.ts b/tests/responses/responses-shadow-intercept.test.ts index 55e866118aa..ed4b81d43e4 100644 --- a/tests/responses/responses-shadow-intercept.test.ts +++ b/tests/responses/responses-shadow-intercept.test.ts @@ -2,7 +2,8 @@ * Shadow call intercept source-model matching (issue #311): Codex 0.145.0 moved * its hard-coded helper model from gpt-5.4-mini to gpt-5.6-luna. The current * default follows modern clients (gpt-6-luna since Codex 0.154.0, with gpt-5.6-luna - * kept for 0.145.0-0.153.x), while sourceModels keeps an escape hatch. + * kept for 0.145.0-0.153.x) and covers gpt-5.6-terra, the model Codex asks for its + * background memory-consolidation pass, while sourceModels keeps an escape hatch. */ import { afterEach, describe, expect, test } from "bun:test"; import { mkdtempSync, readFileSync } from "node:fs"; @@ -38,6 +39,8 @@ describe("isShadowSourceModel", () => { expect(isShadowSourceModel("gpt-6-luna-2026-09")).toBe(true); expect(isShadowSourceModel("gpt-5.6-luna")).toBe(true); expect(isShadowSourceModel("gpt-5.6-luna-2026-08")).toBe(true); + expect(isShadowSourceModel("gpt-5.6-terra")).toBe(true); + expect(isShadowSourceModel("gpt-5.6-terra-2026-08")).toBe(true); }); test("does not match the legacy helper by default but supports an explicit override", () => { @@ -46,7 +49,6 @@ describe("isShadowSourceModel", () => { }); test("does not match non-helper models", () => { - expect(isShadowSourceModel("gpt-5.6-terra")).toBe(false); expect(isShadowSourceModel("gpt-5.5")).toBe(false); expect(isShadowSourceModel("gpt-5.6-sol")).toBe(false); expect(isShadowSourceModel("gpt-6-sol")).toBe(false); @@ -85,7 +87,6 @@ describe("shouldInterceptShadowCall", () => { test("does not intercept non-source models", () => { const source = { providerName: "openai", modelId: "gpt-5.6-luna" }; const target = { providerName: "xai", modelId: "grok-4.5" }; - expect(shouldInterceptShadowCall("gpt-5.6-terra", undefined, source, target)).toBe(false); expect(shouldInterceptShadowCall("gpt-5.5", undefined, source, target)).toBe(false); }); @@ -305,18 +306,39 @@ describe("shadow call intercept request path (issue #311)", () => { expect(logCtx.shadowCallRewrittenFrom).toBe("gpt-5.4-mini"); }); - test("leaves gpt-5.6-terra requests unrewritten", async () => { - let sawFetch = false; - globalThis.fetch = (async () => { - sawFetch = true; - return new Response(JSON.stringify({ error: { message: "unreachable" } }), { status: 500 }); + // gpt-5.6-terra is the model Codex asks for its background memory-consolidation pass. That is + // helper traffic by role, so it is a default source model: an install that intercepts helpers + // must not leave the memory pipeline on the account it routed away from. + test("rewrites a gpt-5.6-terra memory-consolidation call and records that prefix", async () => { + takeSpendHome(); + const bodies: Array> = []; + const logCtx: RequestLogContext = { model: "", provider: "" }; + globalThis.fetch = (async (_url: unknown, init?: RequestInit) => { + bodies.push(JSON.parse(String(init?.body ?? "{}")) as Record); + return chatOk("ok"); + }) as typeof fetch; + + await post(interceptConfig(), "gpt-5.6-terra", "turn", logCtx); + + expect(bodies.length).toBe(1); + expect(String(bodies[0]?.model ?? "")).toContain("grok-4.5"); + expect(logCtx.shadowCallRewrittenFrom).toBe("gpt-5.6-terra"); + }); + + test("a turn tagged x-openai-subagent: memory_consolidation is intercepted too", async () => { + takeSpendHome(); + const bodies: Array> = []; + globalThis.fetch = (async (_url: unknown, init?: RequestInit) => { + bodies.push(JSON.parse(String(init?.body ?? "{}")) as Record); + return chatOk("ok"); }) as typeof fetch; - const response = await post(interceptConfig(), "gpt-5.6-terra"); - // gpt-5.6-terra is not routable in this minimal config: the request must fail - // routing (404) BEFORE any upstream fetch — proving no shadow rewrite happened. - expect(sawFetch).toBe(false); - expect(response.status).toBe(404); + await post(interceptConfig(), "gpt-5.6-terra", undefined, { model: "", provider: "" }, { + "x-openai-subagent": "memory_consolidation", + }); + + expect(bodies.length).toBe(1); + expect(String(bodies[0]?.model ?? "")).toContain("grok-4.5"); }); }); @@ -574,7 +596,7 @@ describe("shadow-call settings API reports the intercepted source models", () => test("GET reports the helper-model defaults, GPT-6 Luna first", async () => { await withTempHome(async () => { const body = await shadowApi({ port: 0, defaultProvider: "xai", providers: {} } as OcxConfig, "GET"); - expect(body.sourceModels).toEqual(["gpt-6-luna", "gpt-5.6-luna"]); + expect(body.sourceModels).toEqual(["gpt-6-luna", "gpt-5.6-luna", "gpt-5.6-terra"]); }); });