From 9d5b6f1399da7e7681b30e274926529537f4630a Mon Sep 17 00:00:00 2001 From: codingbo Date: Sat, 26 Sep 2026 22:08:44 +0800 Subject: [PATCH] fix(providers): calibrate and expand Command Code model reasoning effort ladders Closes #5096 --- .../docs/reference/configuration/providers.md | 8 ++ scripts/test-layout/layout.json | 1 + src/providers/command-code-efforts.ts | 43 ++++++++--- structure/providers-and-adapters.md | 8 ++ tests/fixtures/test-layout-expected.json | 1 + tests/providers/command-code-efforts.test.ts | 75 +++++++++++++++++++ tests/providers/command-code-provider.test.ts | 54 ++++++------- tests/providers/commandcode-provider.test.ts | 10 +-- .../provider-registry-parity.test.ts | 12 +-- 9 files changed, 164 insertions(+), 48 deletions(-) create mode 100644 tests/providers/command-code-efforts.test.ts diff --git a/docs-site/src/content/docs/reference/configuration/providers.md b/docs-site/src/content/docs/reference/configuration/providers.md index e45d68b4bbb..b6facb3a01b 100644 --- a/docs-site/src/content/docs/reference/configuration/providers.md +++ b/docs-site/src/content/docs/reference/configuration/providers.md @@ -292,6 +292,14 @@ Providers can expose a built-in shorthand, such as `agy` for `google-antigravity | `unsafeAllowNativeLocalExec?` | `boolean` | Cursor legacy boolean, equivalent to `nativeLocalExec: "on"` only when the newer field is unset. | | `nativeLocalExec?` | `"off" \| "codex-sandbox" \| "on"` | Cursor local-exec policy. `off` is default; `codex-sandbox` currently fails closed like `off`. | +Command Code's shipped per-model effort defaults include live API measurements. DeepSeek +v4/v4.1 Flash (including v4 Flash Vision), GLM-5.3 and GLM-5.3-Flash, Qwen3.8-Flash, +and Gemini-3.7-Flash support `low`, `medium`, `high`, `xhigh`, and `max`. Ladders remain +model-specific: `poolside/laguna-s-2.1-free` offers only `medium`, while Gemini-3.8-Flash +and MiMo-v2.5-Pro offer `low`, `medium`, and `high`. To override a pinned Command Code +row, set `modelReasoningEffortsAuthoritative: true` together with that model's +`modelReasoningEfforts` list. + Provider registration and replacement (`POST /api/providers`) validate `responsesPath` and `chatCompletionsPath` before changing live configuration or disk state. `PATCH /api/providers?name=` merges the request body with the stored provider; updates touching fields beyond `disabled` — except `requestPacing`-only updates — validate the merged provider's paths the same way before saving, and an invalid retained path returns `400` with the configuration unchanged. The same path rules apply when loading a configuration file. ### What a provider save keeps diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index c10d490a8e3..c9de44cc21d 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -750,6 +750,7 @@ "combo-workspace-data.test.ts": "gui", "combos.test.ts": "codex-integration", "command-code-error-finish.test.ts": "providers", + "command-code-efforts.test.ts": "providers", "command-code-provider.test.ts": "providers", "command-code-quota.test.ts": "providers", "command-code-tool-text.test.ts": "providers", diff --git a/src/providers/command-code-efforts.ts b/src/providers/command-code-efforts.ts index 64919d6d95d..8682327bf77 100644 --- a/src/providers/command-code-efforts.ts +++ b/src/providers/command-code-efforts.ts @@ -22,7 +22,8 @@ import { readBoundedResponseBody } from "../lib/bounded-body"; * `ultra` returned 400. A correction therefore adds the rungs a profile newly lists and keeps the * rungs the upstream measurably accepts; dropping them would strip an effort that works today, * including Codex's default `high`. The refresh path decodes the same payload, so it narrows a row - * only after the upstream actually rejects a rung. + * only after the upstream actually rejects a rung. Issue #5096 supplies additional live API + * measurements that widen the DeepSeek Flash, GLM-5.3, Gemini-3.7-Flash and HY4 rows. */ const COMMAND_CODE_MODEL_EFFORTS = { // Captured profile payload 2026-09-23: claude-fable-5-1.html. @@ -36,11 +37,11 @@ const COMMAND_CODE_MODEL_EFFORTS = { profileUrl: "https://commandcode.ai/models/claude-opus-5-5", }, "deepseek/deepseek-v4-flash": { - efforts: ["high", "max"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/deepseek-v4-flash", }, "deepseek/deepseek-v4-flash-vision-exp": { - efforts: ["high", "max"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/deepseek-v4-flash-vision-exp", }, // Captured profile payload 2026-09-23: deepseek-v4-flash-fast.html. @@ -50,7 +51,7 @@ const COMMAND_CODE_MODEL_EFFORTS = { }, // Captured profile payload 2026-09-23: deepseek-v4-1-flash.html. "deepseek/deepseek-v4.1-flash": { - efforts: ["low", "high", "max"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/deepseek-v4-1-flash", }, "gpt-5.6-luna": { @@ -58,7 +59,7 @@ const COMMAND_CODE_MODEL_EFFORTS = { profileUrl: "https://commandcode.ai/models/gpt-5-6-luna", }, "google/gemini-3.7-flash": { - efforts: ["low", "medium", "high"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/gemini-3-7-flash", }, // Captured profile payload 2026-09-23: gemini-3-8-flash.html. @@ -83,11 +84,11 @@ const COMMAND_CODE_MODEL_EFFORTS = { profileUrl: "https://commandcode.ai/models/glm-5-2-fast", }, "zai-org/GLM-5.3": { - efforts: ["low", "high", "max"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/glm-5-3", }, "z-ai/glm-5.3-flash": { - efforts: ["low", "high", "max"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/glm-5-3-flash", }, // Captured profile payload 2026-09-23: glm-5-3-flashx.html. @@ -143,7 +144,7 @@ const COMMAND_CODE_MODEL_EFFORTS = { }, // Captured profile payload 2026-09-23: hy4-preview.html. "tencent/hy4-preview": { - efforts: ["low", "medium", "high"], + efforts: ["low", "medium", "high", "xhigh", "max"], profileUrl: "https://commandcode.ai/models/hy4-preview", }, // Captured profile payload 2026-09-23: grok-4-7.html. @@ -151,10 +152,31 @@ const COMMAND_CODE_MODEL_EFFORTS = { efforts: ["low", "medium", "high", "xhigh"], profileUrl: "https://commandcode.ai/models/grok-4-7", }, + // Live API measurements supplied in #5096; no verified public profile URL. + "moonshotai/Kimi-K3": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "MiniMaxAI/MiniMax-M3": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "xiaomi/mimo-v2.5": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "xai/grok-4.5": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "xai/grok-4.6": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "tencent/hy3-paid": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "stepfun/Step-3.7-Flash": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "Qwen/Qwen3.8-Max": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "Qwen/Qwen3.8-27B": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "nvidia/nemotron-3-ultra-550b-a55b": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "meituan/LongCat-2.0:free": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "inclusionai/ling-3.0-flash-sante:free": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "thinkingmachines/inkling-small": { efforts: ["low", "medium", "high", "xhigh", "max"] }, + "moonshotai/Kimi-K2.7-Code": { efforts: ["low", "medium", "high", "xhigh"] }, + "moonshotai/Kimi-K2.7-Code-Highspeed": { efforts: ["low", "high", "xhigh", "max"] }, + "xiaomi/mimo-v2.5-pro": { efforts: ["low", "medium", "high"] }, + "Qwen/Qwen3.7-32B": { efforts: ["low", "medium", "high", "xhigh"] }, + "Qwen/Qwen3.7-72B": { efforts: ["low", "medium", "high", "xhigh"] }, + "Qwen/Qwen3.6-35B-A22B": { efforts: ["low", "medium", "high", "xhigh"] }, + "poolside/laguna-s-2.1-free": { efforts: ["medium"] }, } as const; /** - * Official Command Code model-profile facts, not a model catalog. Models remain + * Command Code profile facts and live API measurements (#5096), not a model catalog. Models remain * account-scoped and come exclusively from the authenticated /provider/v1/models endpoint. */ export const COMMAND_CODE_MODEL_REASONING_EFFORTS: Record = Object.fromEntries( @@ -273,7 +295,7 @@ export async function refreshCommandCodeReasoningEfforts( destination = DEFAULT_EFFORT_DESTINATION, ): Promise { const key = cacheKey(modelId, destination); - let profile: { efforts: readonly string[]; profileUrl: string } | undefined; + let profile: { efforts: readonly string[]; profileUrl?: string } | undefined; for (const [id, row] of Object.entries(COMMAND_CODE_MODEL_EFFORTS)) { if (keyFor(id) === keyFor(modelId)) { profile = row; @@ -286,6 +308,7 @@ export async function refreshCommandCodeReasoningEfforts( rejected.add(rejectedEffort); rejectedEfforts.set(key, rejected); } + if (!profile.profileUrl) return undefined; try { const response = await fetchFn(profile.profileUrl, { headers: { Accept: "text/html" }, diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index 9871c973ec4..125dd47d213 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -123,6 +123,14 @@ OAuth presets resolve discovery against the same canonical registry transport as before any adapter-specific transport override, so a stale configured `baseUrl` cannot receive an OAuth bearer token. +Command Code effort defaults in `src/providers/command-code-efforts.ts` combine public-profile +facts with the live API measurements from #5096. Both presets share the exact per-model rows; +`xhigh` is preserved when accepted, and narrow ladders such as Laguna's `medium`-only row remain +narrow. Rows without a verified profile URL still record rejected efforts but skip profile fetching. +An explicit `modelReasoningEffortsAuthoritative` model row overrides the shipped ladder; seeded +rows without that flag do not. `tests/providers/command-code-efforts.test.ts` covers the measured +rows and wire values; `tests/providers/command-code-provider.test.ts` covers operator overrides. + ## TypeSafe JEV decision provider `src/providers/registry/entries-extended.ts` owns the canonical `jev` key preset at diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 22679a1b04f..22938a77333 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -576,6 +576,7 @@ "combo-workspace-data.test.ts": "gui", "combos.test.ts": "codex-integration", "command-code-error-finish.test.ts": "providers", + "command-code-efforts.test.ts": "providers", "command-code-provider.test.ts": "providers", "command-code-quota.test.ts": "providers", "command-code-tool-text.test.ts": "providers", diff --git a/tests/providers/command-code-efforts.test.ts b/tests/providers/command-code-efforts.test.ts new file mode 100644 index 00000000000..c88856afd75 --- /dev/null +++ b/tests/providers/command-code-efforts.test.ts @@ -0,0 +1,75 @@ +import { afterEach, expect, test } from "bun:test"; +import { createCommandCodeAdapter } from "../../src/adapters/command-code"; +import { commandCodeReasoningEfforts, refreshCommandCodeReasoningEfforts, resetCommandCodeReasoningEffortsForTest } from "../../src/providers/command-code-efforts"; +import { PROVIDER_REGISTRY } from "../../src/providers/registry"; +import type { OcxParsedRequest } from "../../src/types"; + +// Issue #5096: measured accepted ladders, including the narrower exceptions. +const measured: Array<[string, string[]]> = [ + ["deepseek/deepseek-v4.1-flash", ["low", "medium", "high", "xhigh", "max"]], + ["deepseek/deepseek-v4-flash", ["low", "medium", "high", "xhigh", "max"]], + ["deepseek/deepseek-v4-flash-vision-exp", ["low", "medium", "high", "xhigh", "max"]], + ["z-ai/glm-5.3-flash", ["low", "medium", "high", "xhigh", "max"]], + ["zai-org/GLM-5.3", ["low", "medium", "high", "xhigh", "max"]], + ["Qwen/Qwen3.8-Flash", ["low", "medium", "high", "xhigh", "max"]], + ["google/gemini-3.7-flash", ["low", "medium", "high", "xhigh", "max"]], + ["moonshotai/Kimi-K3", ["low", "medium", "high", "xhigh", "max"]], + ["MiniMaxAI/MiniMax-M3", ["low", "medium", "high", "xhigh", "max"]], + ["xiaomi/mimo-v2.5", ["low", "medium", "high", "xhigh", "max"]], + ["xai/grok-4.5", ["low", "medium", "high", "xhigh", "max"]], + ["xai/grok-4.6", ["low", "medium", "high", "xhigh", "max"]], + ["tencent/hy3-paid", ["low", "medium", "high", "xhigh", "max"]], + ["tencent/hy4-preview", ["low", "medium", "high", "xhigh", "max"]], + ["stepfun/Step-3.7-Flash", ["low", "medium", "high", "xhigh", "max"]], + ["Qwen/Qwen3.8-Max", ["low", "medium", "high", "xhigh", "max"]], + ["Qwen/Qwen3.8-27B", ["low", "medium", "high", "xhigh", "max"]], + ["meta/muse-spark-1.2-contributor", ["low", "medium", "high", "xhigh", "max"]], + ["meta/muse-spark-1.3-contributor", ["low", "medium", "high", "xhigh", "max"]], + ["nvidia/nemotron-3-ultra-550b-a55b", ["low", "medium", "high", "xhigh", "max"]], + ["meituan/LongCat-2.0:free", ["low", "medium", "high", "xhigh", "max"]], + ["inclusionai/ling-3.0-flash-sante:free", ["low", "medium", "high", "xhigh", "max"]], + ["thinkingmachines/inkling-small", ["low", "medium", "high", "xhigh", "max"]], + ["moonshotai/Kimi-K2.7-Code", ["low", "medium", "high", "xhigh"]], + ["moonshotai/Kimi-K2.7-Code-Highspeed", ["low", "high", "xhigh", "max"]], + ["xiaomi/mimo-v2.5-pro", ["low", "medium", "high"]], + ["Qwen/Qwen3.7-32B", ["low", "medium", "high", "xhigh"]], + ["Qwen/Qwen3.7-72B", ["low", "medium", "high", "xhigh"]], + ["Qwen/Qwen3.6-35B-A22B", ["low", "medium", "high", "xhigh"]], + ["poolside/laguna-s-2.1-free", ["medium"]], + ["google/gemini-3.8-flash", ["low", "medium", "high"]], +]; + +const adapter = createCommandCodeAdapter({ adapter: "command-code", baseUrl: "https://api.commandcode.ai", apiKey: "synthetic-command-key" }); +function request(modelId: string, reasoning: string): OcxParsedRequest { + return { modelId, stream: true, context: { systemPrompt: [], messages: [], tools: [] }, + options: { reasoning, maxOutputTokens: 100 } }; +} +afterEach(() => resetCommandCodeReasoningEffortsForTest()); + +test.each(measured)("%s exposes and forwards every measured effort", async (modelId, ladder) => { + expect(commandCodeReasoningEfforts(modelId)).toEqual(ladder); + expect(commandCodeReasoningEfforts(modelId.toUpperCase())).toEqual(ladder); + for (const id of ["command-code", "commandcode"]) { + expect(PROVIDER_REGISTRY.find(entry => entry.id === id)?.modelReasoningEfforts?.[modelId]).toEqual(ladder); + } + for (const effort of ladder) { + const built = await adapter.buildRequest(request(modelId, effort)); + expect(JSON.parse(built.body).params.reasoning_effort).toBe(effort); + } + for (const effort of ["low", "medium", "high", "xhigh", "max"]) { + if (ladder.includes(effort)) continue; + const built = await adapter.buildRequest(request(modelId, effort)); + expect(JSON.parse(built.body).params).not.toHaveProperty("reasoning_effort"); + } +}); + +test("a measured row without a profile remembers rejections without guessing a URL", async () => { + const model = "moonshotai/Kimi-K3"; + let calls = 0; + const fetch = async () => { calls++; return new Response("", { status: 404 }); }; + expect(await refreshCommandCodeReasoningEfforts(model, fetch, "max")).toBeUndefined(); + expect(calls).toBe(0); + expect(commandCodeReasoningEfforts(model)).toEqual(["low", "medium", "high", "xhigh"]); + const built = await adapter.buildRequest(request(model, "max")); + expect(JSON.parse(built.body).params).not.toHaveProperty("reasoning_effort"); +}); diff --git a/tests/providers/command-code-provider.test.ts b/tests/providers/command-code-provider.test.ts index 126f69649e0..577991a041a 100644 --- a/tests/providers/command-code-provider.test.ts +++ b/tests/providers/command-code-provider.test.ts @@ -110,7 +110,7 @@ describe("Command Code provider", () => { }); expect(registry?.models).toBeUndefined(); expect(registry?.modelReasoningEfforts).toMatchObject({ - "deepseek/deepseek-v4-flash": ["high", "max"], + "deepseek/deepseek-v4-flash": ["low", "medium", "high", "xhigh", "max"], "zai-org/GLM-5.2": ["high", "max"], }); expect(OAUTH_PROVIDERS["command-code"]?.providerConfig).toMatchObject({ @@ -138,7 +138,7 @@ describe("Command Code provider", () => { "zai-org/GLM-5": ["high", "max"], "zai-org/GLM-5.1": ["high", "max"], "zai-org/GLM-5.2-Fast": ["high", "max"], - "zai-org/GLM-5.3": ["low", "high", "max"], + "zai-org/GLM-5.3": ["low", "medium", "high", "xhigh", "max"], }); }); @@ -154,14 +154,14 @@ describe("Command Code provider", () => { const apiKey = PROVIDER_REGISTRY.find(row => row.id === "commandcode"); for (const [label, entry] of [["oauth", oauth], ["api-key", apiKey]] as const) { expect(entry?.modelReasoningEfforts?.["z-ai/glm-5.3-flash"], `${label} preset ladder`) - .toEqual(["low", "high", "max"]); + .toEqual(["low", "medium", "high", "xhigh", "max"]); } // Distinct rows for distinct upstream models: GLM-5.3 and GLM-5.3-Flash happen to // share a ladder today, but neither may be derived from the other. - expect(commandCodeReasoningEfforts("z-ai/glm-5.3-flash")).toEqual(["low", "high", "max"]); - expect(commandCodeReasoningEfforts("zai-org/GLM-5.3")).toEqual(["low", "high", "max"]); + expect(commandCodeReasoningEfforts("z-ai/glm-5.3-flash")).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(commandCodeReasoningEfforts("zai-org/GLM-5.3")).toEqual(["low", "medium", "high", "xhigh", "max"]); // The reported id arrives lowercase from live discovery; a caller may still fold case. - expect(commandCodeReasoningEfforts("Z-AI/GLM-5.3-Flash")).toEqual(["low", "high", "max"]); + expect(commandCodeReasoningEfforts("Z-AI/GLM-5.3-Flash")).toEqual(["low", "medium", "high", "xhigh", "max"]); // Nothing widened into a substring match: a sibling that upstream does not list // must stay unknown rather than inheriting the Flash ladder. expect(commandCodeReasoningEfforts("z-ai/glm-5.3-flash-vision")).toBeUndefined(); @@ -179,11 +179,11 @@ describe("Command Code provider", () => { const apiKey = PROVIDER_REGISTRY.find(row => row.id === "commandcode"); for (const [label, entry] of [["oauth", oauth], ["api-key", apiKey]] as const) { expect(entry?.modelReasoningEfforts?.["deepseek/deepseek-v4.1-flash"], `${label} preset ladder`) - .toEqual(["low", "high", "max"]); + .toEqual(["low", "medium", "high", "xhigh", "max"]); expect(entry?.modelReasoningEfforts?.["Qwen/Qwen3.8-Flash"], `${label} preset ladder`) .toEqual(["low", "medium", "high", "xhigh", "max"]); } - expect(commandCodeReasoningEfforts("deepseek/deepseek-v4.1-flash")).toEqual(["low", "high", "max"]); + expect(commandCodeReasoningEfforts("deepseek/deepseek-v4.1-flash")).toEqual(["low", "medium", "high", "xhigh", "max"]); expect(commandCodeReasoningEfforts("Qwen/Qwen3.8-Flash")).toEqual(["low", "medium", "high", "xhigh", "max"]); // The live-discovered id may arrive in any case; the lookup folds it. expect(commandCodeReasoningEfforts("qwen/qwen3.8-flash")).toEqual(["low", "medium", "high", "xhigh", "max"]); @@ -693,7 +693,7 @@ describe("Command Code provider", () => { }); test("does not advertise an unverified effort for models absent from the official table", async () => { - const built = await builtRequest(parsed("moonshotai/Kimi-K3")); + const built = await builtRequest(parsed("unknown/unmeasured-model")); expect(JSON.parse(built.body).params).not.toHaveProperty("reasoning_effort"); }); @@ -743,7 +743,7 @@ describe("Command Code provider", () => { options: { reasoning: "ultra", maxOutputTokens: 100 }, }); expect(JSON.parse(ultra.body).params).not.toHaveProperty("reasoning_effort"); - // Deepseek/glm still alias xhigh/ultra→max per their official profiles. + // DeepSeek retains the ultra alias but forwards its accepted xhigh rung unchanged. const deepseekUltra = await builtRequest({ ...parsed("deepseek/deepseek-v4-flash"), options: { reasoning: "ultra", maxOutputTokens: 100 }, @@ -753,14 +753,14 @@ describe("Command Code provider", () => { ...parsed("deepseek/deepseek-v4-flash"), options: { reasoning: "xhigh", maxOutputTokens: 100 }, }); - expect(JSON.parse(deepseekXhigh.body).params.reasoning_effort).toBe("max"); + expect(JSON.parse(deepseekXhigh.body).params.reasoning_effort).toBe("xhigh"); }); - test("maps ultra and xhigh to the max wire effort and honors legacy alias ids", async () => { + test("maps ultra to max, preserves xhigh, and honors legacy alias ids", async () => { const ultra = await builtRequest({ ...parsed(), options: { reasoning: "ultra", maxOutputTokens: 100 } }); expect(JSON.parse(ultra.body).params.reasoning_effort).toBe("max"); const xhigh = await builtRequest({ ...parsed(), options: { reasoning: "xhigh", maxOutputTokens: 100 } }); - expect(JSON.parse(xhigh.body).params.reasoning_effort).toBe("max"); + expect(JSON.parse(xhigh.body).params.reasoning_effort).toBe("xhigh"); // Legacy compatibility id resolves to the canonical effort table before the lookup. const legacy = await builtRequest({ ...parsed(), modelId: "deepseek-v4-flash" }); expect(JSON.parse(legacy.body).params.reasoning_effort).toBe("high"); @@ -843,7 +843,7 @@ describe("Command Code provider", () => { if (mode === "prepaid") expect(await response.text()).toContain("unsupported reasoning_effort"); else expect(JSON.parse(generated[1]!.body!).params).not.toHaveProperty("reasoning_effort"); } finally { dispose(); } - expect(commandCodeReasoningEfforts("deepseek/deepseek-v4-flash")).toEqual(["high"]); + expect(commandCodeReasoningEfforts("deepseek/deepseek-v4-flash")).toEqual(["low", "medium", "high", "xhigh"]); }); /* @@ -862,10 +862,10 @@ describe("Command Code provider", () => { * `modelReasoningEffortsAuthoritative` is never written by seeding, so its presence does. */ test("an authoritative operator ladder reaches the wire", async () => { - // Shipped: deepseek/deepseek-v4.1-flash is ["low", "high", "max"], so xhigh aliases to max. - expect(commandCodeReasoningEfforts("deepseek/deepseek-v4.1-flash")).toEqual(["low", "high", "max"]); + // Shipped: deepseek/deepseek-v4-flash-fast is ["low", "high", "max"], so xhigh aliases to max. + expect(commandCodeReasoningEfforts("deepseek/deepseek-v4-flash-fast")).toEqual(["low", "high", "max"]); const shipped = await builtRequest({ - ...parsed("deepseek/deepseek-v4.1-flash"), + ...parsed("deepseek/deepseek-v4-flash-fast"), options: { reasoning: "xhigh", maxOutputTokens: 100 }, }); expect(JSON.parse(shipped.body).params.reasoning_effort).toBe("max"); @@ -873,10 +873,10 @@ describe("Command Code provider", () => { const widened = createCommandCodeAdapter({ ...provider, modelReasoningEffortsAuthoritative: true, - modelReasoningEfforts: { "deepseek/deepseek-v4.1-flash": ["low", "medium", "high", "xhigh", "max"] }, + modelReasoningEfforts: { "deepseek/deepseek-v4-flash-fast": ["low", "medium", "high", "xhigh", "max"] }, } as OcxProviderConfig); const built = await widened.buildRequest({ - ...parsed("deepseek/deepseek-v4.1-flash"), + ...parsed("deepseek/deepseek-v4-flash-fast"), options: { reasoning: "xhigh", maxOutputTokens: 100 }, }); expect(JSON.parse(built.body).params.reasoning_effort).toBe("xhigh"); @@ -885,10 +885,10 @@ describe("Command Code provider", () => { const narrowed = createCommandCodeAdapter({ ...provider, modelReasoningEffortsAuthoritative: true, - modelReasoningEfforts: { "deepseek/deepseek-v4.1-flash": ["high"] }, + modelReasoningEfforts: { "deepseek/deepseek-v4-flash-fast": ["high"] }, } as OcxProviderConfig); const stripped = await narrowed.buildRequest({ - ...parsed("deepseek/deepseek-v4.1-flash"), + ...parsed("deepseek/deepseek-v4-flash-fast"), options: { reasoning: "max", maxOutputTokens: 100 }, }); expect(JSON.parse(stripped.body).params).not.toHaveProperty("reasoning_effort"); @@ -1054,18 +1054,18 @@ describe("Command Code provider", () => { const modelId = "deepseek/deepseek-v4-flash"; const alternate = "https://alternate.example/command-code"; const fetch = (async () => new Response("Reasoning efforts high, max are supported; no mapping.")) as typeof globalThis.fetch; - expect(await refreshCommandCodeReasoningEfforts(modelId, fetch, "max", provider.baseUrl)).toEqual(["high"]); - expect(commandCodeReasoningEfforts(modelId, alternate)).toEqual(["high", "max"]); + expect(await refreshCommandCodeReasoningEfforts(modelId, fetch, "max", provider.baseUrl)).toEqual(["low", "medium", "high", "xhigh"]); + expect(commandCodeReasoningEfforts(modelId, alternate)).toEqual(["low", "medium", "high", "xhigh", "max"]); const options = { reasoning: "max", maxOutputTokens: 100 }; const officialRequest = await createCommandCodeAdapter(provider).buildRequest({ ...parsed(modelId), options }); const alternateRequest = await createCommandCodeAdapter({ ...provider, baseUrl: alternate }).buildRequest({ ...parsed(modelId), options }); expect(JSON.parse(officialRequest.body).params).not.toHaveProperty("reasoning_effort"); expect(JSON.parse(alternateRequest.body).params.reasoning_effort).toBe("max"); - expect(await refreshCommandCodeReasoningEfforts(modelId, fetch, "high", alternate)).toEqual(["max"]); - expect(commandCodeReasoningEfforts(modelId, provider.baseUrl)).toEqual(["high"]); + expect(await refreshCommandCodeReasoningEfforts(modelId, fetch, "high", alternate)).toEqual(["low", "medium", "xhigh", "max"]); + expect(commandCodeReasoningEfforts(modelId, provider.baseUrl)).toEqual(["low", "medium", "high", "xhigh"]); }); - test("uses the 2026-09-23 profile ladders for newly cataloged models", async () => { + test("uses profile ladders plus measured corrections for newly cataloged models", async () => { const cases: Array<[string, string[]]> = [ ["claude-fable-5-1", ["low", "medium", "high", "xhigh", "max"]], ["claude-opus-5-5", ["low", "medium", "high", "xhigh", "max"]], @@ -1074,7 +1074,7 @@ describe("Command Code provider", () => { ["Qwen/Qwen3.8-Omni-Flash", ["low", "medium", "xhigh"]], ["Qwen/Qwen3.8-Max-0902", ["low", "medium", "xhigh"]], ["stepfun/Step-5-Preview", ["low", "medium", "high"]], - ["tencent/hy4-preview", ["low", "medium", "high"]], + ["tencent/hy4-preview", ["low", "medium", "high", "xhigh", "max"]], ["google/gemini-3.8-flash", ["low", "medium", "high"]], ["xai/grok-4.7", ["low", "medium", "high", "xhigh"]], ]; diff --git a/tests/providers/commandcode-provider.test.ts b/tests/providers/commandcode-provider.test.ts index b75324c973c..5061a08c64b 100644 --- a/tests/providers/commandcode-provider.test.ts +++ b/tests/providers/commandcode-provider.test.ts @@ -71,9 +71,9 @@ describe("Command Code provider", () => { apiKeyValidation: "unknown", reasoningEfforts: [], modelReasoningEfforts: { - "deepseek/deepseek-v4-flash-vision-exp": ["high", "max"], + "deepseek/deepseek-v4-flash-vision-exp": ["low", "medium", "high", "xhigh", "max"], "gpt-5.6-luna": ["low", "medium", "high", "xhigh", "max"], - "google/gemini-3.7-flash": ["low", "medium", "high"], + "google/gemini-3.7-flash": ["low", "medium", "high", "xhigh", "max"], }, modelDiscovery: { path: "models", @@ -216,7 +216,7 @@ describe("Command Code provider", () => { expect(deepseek.contextWindow).toBe(1_000_000); expect(deepseek.owned_by).toBe("command-code"); // #1800: discovered models now surface the curated effort table (command-code-efforts.ts). - expect(deepseek.reasoningEfforts).toEqual(["high", "max"]); + expect(deepseek.reasoningEfforts).toEqual(["low", "medium", "high", "xhigh", "max"]); const haiku = models.find(row => row.id === "claude-haiku-4-5-20251001")!; expect(haiku.contextWindow).toBe(200_000); @@ -228,7 +228,7 @@ describe("Command Code provider", () => { expect(models.find(row => row.id === "deepseek/deepseek-v4-flash-vision-exp")) .toMatchObject({ id: "deepseek/deepseek-v4-flash-vision-exp", - reasoningEfforts: ["high", "max"], + reasoningEfforts: ["low", "medium", "high", "xhigh", "max"], }); expect(models.find(row => row.id === "gpt-5.6-luna")).toMatchObject({ id: "gpt-5.6-luna", @@ -236,7 +236,7 @@ describe("Command Code provider", () => { }); expect(models.find(row => row.id === "google/gemini-3.7-flash")).toMatchObject({ id: "google/gemini-3.7-flash", - reasoningEfforts: ["low", "medium", "high"], + reasoningEfforts: ["low", "medium", "high", "xhigh", "max"], }); expect(models.find(row => row.id === "Qwen/Qwen3.8-Flash")?.reasoningEfforts) .toEqual(["low", "medium", "high", "xhigh", "max"]); diff --git a/tests/providers/provider-registry-parity.test.ts b/tests/providers/provider-registry-parity.test.ts index 36ab8d2b5c4..d6b75c5565f 100644 --- a/tests/providers/provider-registry-parity.test.ts +++ b/tests/providers/provider-registry-parity.test.ts @@ -1652,7 +1652,7 @@ describe("provider registry parity", () => { provider: "commandcode", }); expect(model.id).toBe("z-ai/glm-5.3-flash"); - expect(model.reasoningEfforts).toEqual(["low", "high", "max"]); + expect(model.reasoningEfforts).toEqual(["low", "medium", "high", "xhigh", "max"]); const entries = buildCatalogEntries(nativeTemplate() as never, [], [model]); const entry = entries.find(e => e.slug === "commandcode/z-ai-glm-5.3-flash"); @@ -1661,7 +1661,7 @@ describe("provider registry parity", () => { expect(entry?.supported_reasoning_levels).not.toEqual([]); // Routed catalogs append the synthetic top rung, as every other routed row above does. expect((entry?.supported_reasoning_levels as { effort: string }[]).map(l => l.effort)) - .toEqual(["low", "high", "max", "ultra"]); + .toEqual(["low", "medium", "high", "xhigh", "max", "ultra"]); }); /* * #1043. Zen publishes no modality metadata, so the classification below is an @@ -1856,8 +1856,8 @@ describe("renamed fixed-key destination reasoning metadata", () => { test("fills known model tables and unknown-model default for CommandCode", () => { const provider = make(); enrichProviderFromRegistry("CommandCode", provider); - expect(configuredReasoningEfforts(provider, known)).toEqual(["high", "max"]); - expect(configuredReasoningEfforts(provider, newer)).toEqual(["low", "high", "max"]); + expect(configuredReasoningEfforts(provider, known)).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(configuredReasoningEfforts(provider, newer)).toEqual(["low", "medium", "high", "xhigh", "max"]); expect(configuredReasoningEfforts(provider, "unknown-model")).toEqual([]); }); test("preserves explicit entries and clones arrays without losing other table rows", () => { @@ -1870,7 +1870,7 @@ describe("renamed fixed-key destination reasoning metadata", () => { enrichProviderFromRegistry("CommandCode", provider); expect(provider).toEqual(once); expect(configuredReasoningEfforts(provider, known)).toEqual(["low"]); - expect(configuredReasoningEfforts(provider, newer)).toEqual(["low", "high", "max"]); + expect(configuredReasoningEfforts(provider, newer)).toEqual(["low", "medium", "high", "xhigh", "max"]); expect(configuredReasoningEfforts(provider, "custom")).toEqual([]); expect(configuredReasoningEfforts(provider, "unknown-model")).toEqual(["medium"]); provider.modelReasoningEfforts![known]!.push("high"); @@ -1882,7 +1882,7 @@ describe("renamed fixed-key destination reasoning metadata", () => { const provider = make({ modelReasoningEfforts: { [known]: [] } }); enrichProviderFromRegistry("CommandCode", provider); expect(configuredReasoningEfforts(provider, known)).toEqual([]); - expect(configuredReasoningEfforts(provider, newer)).toEqual(["low", "high", "max"]); + expect(configuredReasoningEfforts(provider, newer)).toEqual(["low", "medium", "high", "xhigh", "max"]); }); test("does not infer metadata for a different adapter, OAuth, or unrelated endpoint", () => { for (const override of [{ adapter: "openai-responses" }, { authMode: "oauth" as const }, { baseUrl: "https://example.test/v1" }]) {