diff --git a/docs-site/src/content/docs/reference/adapters.md b/docs-site/src/content/docs/reference/adapters.md index 5f15be73148..335d877a5a6 100644 --- a/docs-site/src/content/docs/reference/adapters.md +++ b/docs-site/src/content/docs/reference/adapters.md @@ -571,8 +571,10 @@ configuration that names the old id is rewritten at startup. session token doubled and dash-joined in an `Authorization: Basic` header while the protobuf body keeps one copy, the request envelope goes up uncompressed, and `Metadata` #31 carries a 732-character device fingerprint whose length — not value — the service checks. Inside - `CompletionConfiguration`, #2 is the output cap and #3 is the context window; swapping those two - makes every turn fail with an opaque `invalid_argument`. A temperature of exactly 0 is refused, so + `CompletionConfiguration`, #2 is the output cap and #3 is `max_newlines`, sent at a fixed large + value; swapping those two makes every turn fail with an opaque `invalid_argument`. With no caller + or configured cap, the output cap is the selected model's own ceiling from the catalog, and 8192 + only when the catalog is unavailable. A temperature of exactly 0 is refused, so it is clamped to the smallest accepted value. - A pre-output 429 with a stated recovery delay is surfaced immediately by default, releasing the admitted turn's shared capacity. Set `OPENCODEX_DEVIN_STATED_RESET_WAIT_MS` to a positive cumulative @@ -591,13 +593,18 @@ configuration that names the old id is rewritten at startup. - Experimental unofficial bridge; not shown in the dashboard preset by default. See the [provider guide](/guides/providers/) for login instructions. -For SWE-2, an explicit reasoning effort overrides an effort suffix in the model -id. For example, `swe-2-high` with `medium` selects the native `swe-2-medium` UID; -`xhigh`, `ultra`, and `max` select `swe-2-max`. Values below Medium select Medium -and do not disable SWE-2 reasoning. Without an explicit effort, a suffixed model -id is preserved. This applies through the shared adapter to every Devin account, -whichever login path minted the credential; other model families keep their -existing suffix precedence. +The model id you pick is a model family, and the reasoning effort picks the +variant. With no effort, the family's own default variant is used: `swe-1-7` +selects `swe-1-7-medium` and `swe-2` selects `swe-2-high`. An effort changes only +the effort and lands on the lowest variant at or above it, or the highest one +below when nothing is above, so a missing rung never turns reasoning down: +`swe-1-7` at `high` selects the Max row `swe-1-7`, `swe-2` at `low` selects +`swe-2-medium`, and `kimi-k3` at `medium` selects `kimi-k3-high`. `fast` selects the Fast variant and `1m`, `max-1m` or `none-1m` +the 1M-context variant where the family has one; otherwise they change nothing. +A suffixed id such as `claude-opus-5-high-fast` keeps its variant without an +effort, and with one keeps its Fast and context settings. Resolution never moves +to another family. When the account catalog is unavailable, the effort is +appended to the id instead. ## `azure-openai` (alias: `azure`) diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index f3f16fae58b..aee7f953b46 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -911,6 +911,7 @@ "devin-adapter.test.ts": "providers", "devin-cli-authmode-migration.test.ts": "providers", "devin-effort-ladder.test.ts": "providers", + "devin-family-resolution.test.ts": "providers", "devin-hardening.test.ts": "providers", "devin-image-passthrough.test.ts": "providers", "devin-live-models.test.ts": "providers", diff --git a/src/adapters/devin.ts b/src/adapters/devin.ts index 6c7c8e3ff6d..1412757778b 100644 --- a/src/adapters/devin.ts +++ b/src/adapters/devin.ts @@ -11,8 +11,8 @@ import { namespacedToolName } from "../types"; import type { IncomingMeta, ProviderAdapter } from "./base"; import { streamChatEventsWithResetRetry, devinStatedResetWaitMs, allocateCascadeId, CloudChatError, type ChatHistoryItem, type ToolDef } from "./devin/cloud-direct"; import type { ContentPart } from "./devin/cloud-direct/chat"; -import { getCachedCatalog, type CacheEntry } from "./devin/cloud-direct/catalog"; -import { collapseDevinModelUid } from "./devin/live-models"; +import { getCachedCatalog, type CacheEntry, type ModelCatalogEntry } from "./devin/cloud-direct/catalog"; +import { collapseDevinModelUid, devinFamiliesOf, devinFamilyBaseId, selectDevinFamilyMember, type DevinVariantRequest } from "./devin/live-models"; import { buildNonOpenAIToolCatalogNudgeForTools } from "./tool-catalog-nudge"; import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiServer } from "../oauth/devin"; import { isProviderIssuedThinkingSignature } from "../responses/reasoning-envelope"; @@ -131,7 +131,9 @@ const SWE2_EFFORT: Record = { /** * Resolve an explicit effort onto a SWE-2 lane, or undefined when this is not a - * SWE-2 id or the caller named no usable effort. Undefined leaves every existing + * SWE-2 id or the caller named no usable effort. Reached only when the catalog + * is unavailable or lacks family metadata; otherwise the catalog's SWE-2 family + * rows decide, with the same rounding. Undefined leaves every existing * path untouched, which is what keeps other model families on suffix precedence. */ function resolveSwe2Variant(modelId: string, reasoningEffort?: string): string | undefined { @@ -140,17 +142,85 @@ function resolveSwe2Variant(modelId: string, reasoningEffort?: string): string | return mapped ? `swe-2-${mapped}` : undefined; } +const VARIANT_EFFORT_TOKENS = new Set(["none", "minimal", "low", "medium", "high", "xhigh", "max"]); +const EFFORT_ALIASES: Record = { off: "none", ultra: "max" }; + /** - * Resolve the wire model UID using the live catalog as the source of truth. - * Cognition's catalog lists most models with an effort suffix - * (e.g. `gpt-5-6-sol-high`); the base id alone is not accepted for those. + * Read hyphen-separated variant tokens as a request. Any token outside the + * vocabulary voids the whole value, so an unknown effort changes nothing + * rather than half-applying. `priority` is the uid spelling of Fast Mode on + * the GPT rows; a caller never sends it as effort (see CALLER_EFFORT_VALUES). + */ +function variantRequestOf(tokens: string[], allowPriority: boolean): DevinVariantRequest | undefined { + const request: DevinVariantRequest = {}; + for (const raw of tokens) { + const token = EFFORT_ALIASES[raw] ?? raw; + if (VARIANT_EFFORT_TOKENS.has(token)) request.effort = token; + else if (token === "fast" || (allowPriority && token === "priority")) request.fast = true; + else if (token === "1m") request.longContext = true; + else return undefined; + } + return request; +} + +function callerVariantRequest(reasoningEffort: string | undefined): DevinVariantRequest | undefined { + const value = reasoningEffort?.toLowerCase(); + if (!value) return undefined; + const aliased = EFFORT_ALIASES[value] ?? value; + if (!CALLER_EFFORT_VALUES.has(aliased) && aliased !== "minimal") return undefined; + return variantRequestOf(aliased.split("-"), false); +} + +/** + * Resolve through the catalog's family metadata (ClientModelConfig #23/#30/#31). + * Returns undefined when the id belongs to no family, which leaves the + * suffix-based path for legacy rows and catalogs without that metadata. * - * If the catalog is available: use the exact UID when it exists, otherwise - * append the reasoning effort (or `medium` default) and pick a variant the - * account actually has. + * The family id wins over an identical row uid on purpose: bare `swe-1-7` is + * the Max row while the family default is `swe-1-7-medium`, and `glm-5-2` is + * both the family id and its default row. A caller naming the family means the + * family, so no effort selects the default member and an effort moves only the + * effort axis. A caller naming a member row keeps that row's other axes. + */ +function resolveFamilyUid(catalog: CacheEntry, modelId: string, reasoningEffort?: string): string | undefined { + const families = devinFamiliesOf(catalog); + const caller = callerVariantRequest(reasoningEffort); + let members = families.get(modelId); + let anchor: ModelCatalogEntry | undefined; + let fromId: DevinVariantRequest = {}; + if (!members) { + const row = catalog.byUid.get(modelId); + if (row?.familyUid) { + members = families.get(devinFamilyBaseId(row.familyUid)); + if (members) anchor = row; + } + } + if (!members) { + // A suffixed spelling the catalog does not list (`swe-1-7-high`) still + // names a family; its suffix is the request. + const base = collapseDevinModelUid(modelId); + const suffix = base !== modelId ? variantRequestOf(modelId.slice(base.length + 1).split("-"), true) : undefined; + members = suffix ? families.get(base) : undefined; + if (suffix) fromId = suffix; + } + if (!members) return undefined; + if (anchor && !caller) return anchor.modelUid; + const request: DevinVariantRequest = { + ...fromId, + ...(caller?.effort ? { effort: caller.effort } : {}), + ...(caller?.fast || fromId.fast ? { fast: true } : {}), + ...(caller?.longContext || fromId.longContext ? { longContext: true } : {}), + }; + return selectDevinFamilyMember(members, request, anchor)?.modelUid; +} + +/** + * Resolve the wire model UID using the live catalog as the source of truth. * - * If the catalog is unavailable (degraded mode): append the effort suffix - * for any base id that doesn't already carry one, mirroring the catalog shape. + * With family metadata in the catalog, resolution picks a family member by + * axis (see resolveFamilyUid). Without it — legacy rows, an older catalog, or + * no catalog at all — the id is resolved by suffix: an exact or suffixed UID is + * kept, otherwise the reasoning effort (or `medium`) is appended. */ async function resolveWireModelUid( rawModelId: string, @@ -160,28 +230,33 @@ async function resolveWireModelUid( catalog?: CacheEntry | null, ): Promise { const modelId = normalizeDevinModelId(rawModelId); + // Callers that already read the catalog this turn pass it in; an explicit + // null records a failed lookup and must not trigger a same-turn retry — + // failures are not cached, so re-reading would only pay another timeout. + const entry = catalog !== undefined ? catalog : await getCachedCatalog(apiKey, host); + if (entry) { + const fromFamily = resolveFamilyUid(entry, modelId, reasoningEffort); + if (fromFamily) return fromFamily; + } // Explicit effort wins over a suffix the picker already baked into the id, so // `swe-2-high` asked for at `medium` becomes `swe-2-medium` instead of ignoring // the caller. Runs before the shortcut below, which would otherwise return early. const swe2 = resolveSwe2Variant(modelId, reasoningEffort); if (swe2) return swe2; if (hasEffortSuffix(modelId)) return modelId; - // Callers that already read the catalog this turn pass it in; an explicit - // null records a failed lookup and must not trigger a same-turn retry — - // failures are not cached, so re-reading would only pay another timeout. - const entry = catalog !== undefined ? catalog : await getCachedCatalog(apiKey, host); + const effort = reasoningEffort && CALLER_EFFORT_VALUES.has(reasoningEffort) ? reasoningEffort : "medium"; if (entry) { if (entry.byUid.has(modelId)) return modelId; - const effort = reasoningEffort && CALLER_EFFORT_VALUES.has(reasoningEffort) ? reasoningEffort : "medium"; const suffixed = `${modelId}-${effort}`; if (entry.byUid.has(suffixed)) return suffixed; - // Fall back to any enabled variant of this base model. - for (const uid of entry.byUid.keys()) { - if (uid.startsWith(modelId + "-") && !entry.byUid.get(uid)?.disabled) return uid; + // Any enabled variant of this exact base. The base must match after + // collapsing, not as a string prefix: `claude-opus-5-` prefixes + // `claude-opus-5-5-low` and `gpt-5-4-` prefixes `gpt-5-4-mini-low`. + for (const [uid, row] of entry.byUid) { + if (!row.disabled && uid !== modelId && collapseDevinModelUid(uid) === modelId) return uid; } } // Degraded mode: append the default effort suffix. - const effort = reasoningEffort && CALLER_EFFORT_VALUES.has(reasoningEffort) ? reasoningEffort : "medium"; return `${modelId}-${effort}`; } @@ -198,7 +273,9 @@ const positiveTokenCount = (value: unknown): number | undefined => /** * Read a per-model token count for the exact UID selected for this turn. * - * Tries the selected UID and then its collapsed base id, preferring the + * Tries the selected UID, then its catalog family id (the picker id, e.g. + * `claude-sonnet-4-6` for `claude-sonnet-4-6-thinking`), then its collapsed + * base id, preferring the * canonical spelling and accepting dotted or case-folded saved hints — the same * normalization the inference request applies to the model id. Where several * spellings match one id, the smallest wins: a ceiling stated twice is @@ -207,9 +284,10 @@ const positiveTokenCount = (value: unknown): number | undefined => function devinModelTokenHint( record: Record | undefined, modelUid: string, + familyBase?: string, ): number | undefined { if (!record) return undefined; - for (const id of [modelUid, collapseDevinModelUid(modelUid)]) { + for (const id of [modelUid, ...(familyBase ? [familyBase] : []), collapseDevinModelUid(modelUid)]) { const exact = Object.hasOwn(record, id) ? positiveTokenCount(record[id]) : undefined; if (exact !== undefined) return exact; const matches = Object.entries(record) @@ -221,26 +299,6 @@ function devinModelTokenHint( return undefined; } -/** - * Resolve the INPUT ceiling for the exact UID selected for this turn. Catalog - * ClientModelConfig #18 and CompletionConfiguration #3 both carry input tokens; - * the independent output cap is not subtracted here. Smaller operator hints - * cap live evidence, never enlarge it. No evidence leaves the encoder's 128k - * fallback intact; an unrelated or opt-in long-context variant is not evidence. - */ -function resolveDevinMaxInputTokens( - provider: OcxProviderConfig, - modelUid: string, - liveWindow?: number, -): number | undefined { - const contextHint = devinModelTokenHint(provider.modelContextWindows, modelUid) - ?? positiveTokenCount(provider.contextWindow); - const inputHint = devinModelTokenHint(provider.modelMaxInputTokens, modelUid); - const ceilings = [positiveTokenCount(liveWindow), contextHint, inputHint] - .filter((value): value is number => value !== undefined); - return ceilings.length > 0 ? Math.min(...ceilings) : undefined; -} - /** * Resolve the OUTPUT ceiling for this turn, highest authority first: * @@ -248,16 +306,17 @@ function resolveDevinMaxInputTokens( * explicit cap is a request, so a small one is never widened into a * configured larger one; * 2. the configured per-model cap (`modelMaxOutputTokens`), read through the - * same UID-aware hint lookup the input ceiling uses; + * through the UID-aware hint lookup above; * 3. the provider-wide `defaultMaxOutputTokens`; - * 4. undefined, which leaves the cloud-direct encoder's own 8192 fallback in - * place for a provider that configured nothing. + * 4. the catalog's own ceiling for the selected UID (ModelInfo #13) — without + * it every uncapped turn stopped at the encoder's 8192, far below the + * 128k most rows advertise; + * 5. undefined, which leaves the cloud-direct encoder's 8192 fallback in + * place when there is no catalog and nothing configured. * - * This is NOT the history ceiling, and the two must not collapse into one - * number. CompletionConfiguration #2 is the output cap and #3 is the context - * window, so feeding a context window into this resolver would ask Cognition to - * generate a whole window's worth of output. Nothing here reads - * `contextWindow` or `modelContextWindows` for that reason. + * A context window is not an output cap: feeding one into CompletionConfiguration + * #2 would ask Cognition to generate a whole window's worth of output. Nothing + * here reads `contextWindow` or `modelContextWindows` for that reason. * * Step 1 keeps the caller's raw value rather than `positiveTokenCount`: the * inbound parser owns what a caller may send, and re-filtering here would @@ -268,14 +327,16 @@ function resolveDevinMaxOutputTokens( provider: OcxProviderConfig, modelUid: string, requested: number | undefined, + catalogRow?: Pick, ): number | undefined { if (typeof requested === "number") return requested; - return devinModelTokenHint(provider.modelMaxOutputTokens, modelUid) - ?? positiveTokenCount(provider.defaultMaxOutputTokens); + const familyBase = catalogRow?.familyUid ? devinFamilyBaseId(catalogRow.familyUid) : undefined; + return devinModelTokenHint(provider.modelMaxOutputTokens, modelUid, familyBase) + ?? positiveTokenCount(provider.defaultMaxOutputTokens) + ?? positiveTokenCount(catalogRow?.maxOutputTokens); } -/** Pure test seams; runtime uses the same resolvers immediately before dispatch. */ -export const resolveDevinMaxInputTokensForTests = resolveDevinMaxInputTokens; +/** Pure test seam; runtime uses the same resolver immediately before dispatch. */ export const resolveDevinMaxOutputTokensForTests = resolveDevinMaxOutputTokens; export class DevinMissingCredentialError extends Error { @@ -613,8 +674,8 @@ export function createDevinAdapter( // entry: an EU or FedStart account that used provider.baseUrl would send // every RPC to the US server it is not provisioned on. const host = resolveDevinApiServer(provider.baseUrl, credentialProviderId, apiKey); - // One catalog read per turn serves model-UID resolution, the input - // ceiling, and the chat pre-flight inside streamChatEvents. Failures are + // One catalog read per turn serves model-UID resolution, the output + // cap, and the chat pre-flight inside streamChatEvents. Failures are // not cached, so a second read would only pay another fetch timeout on // an otherwise valid turn. const catalog = await getCachedCatalog(apiKey, host, incoming.abortSignal); @@ -636,11 +697,8 @@ export function createDevinAdapter( try { // Read the selected UID's catalog row, not the picker's collapsed base. - const maxInputTokens = resolveDevinMaxInputTokens( - provider, modelUid, catalog?.byUid.get(modelUid)?.contextWindow, - ); const maxOutputTokens = resolveDevinMaxOutputTokens( - provider, modelUid, parsed.options.maxOutputTokens, + provider, modelUid, parsed.options.maxOutputTokens, catalog?.byUid.get(modelUid), ); // A combo child has not committed an outer response yet. Holding its preflight through // a reset wait would also hold the next-target fallback with no client keepalive. @@ -656,10 +714,7 @@ export function createDevinAdapter( messages: mapOcxMessagesToDevin(parsed), tools: mapOcxToolsToDevin(parsed.context.tools), cascadeId, - // Input and output ceilings are separate wire fields. Omitting the - // input hint used to force every model through the 128k default. completionOpts: { - ...(maxInputTokens !== undefined ? { maxInputTokens } : {}), ...(maxOutputTokens !== undefined ? { maxOutputTokens } : {}), ...(typeof parsed.options.temperature === "number" ? { temperature: parsed.options.temperature } : {}), ...(typeof parsed.options.topP === "number" ? { topP: parsed.options.topP } : {}), diff --git a/src/adapters/devin/cloud-direct/catalog.ts b/src/adapters/devin/cloud-direct/catalog.ts index 1f45027c90b..ef708f6b97a 100644 --- a/src/adapters/devin/cloud-direct/catalog.ts +++ b/src/adapters/devin/cloud-direct/catalog.ts @@ -35,8 +35,18 @@ * #5 supports_images bool ← tri-state: absent stays unknown * #18 max_input_tokens varint ← per-account context window * #22 model_uid string ← what `GetChatMessage` accepts + * #23 model_info ModelInfo { #6 features { #15 supports_thinking }, + * #13 max_output_tokens, #23 model_family_uid } + * #30 family_metadata ModelFamilyMetadata { #1 family_label, + * #2 repeated Entry { #1 axis key, #2 { #1 order, #2 name } }, + * #3 is_default_model_in_family } + * #31 is_default_model_in_family bool * } * + * #23/#30/#31 were read back from a live 267-row catalog (2026-09-27): every + * current row names its family, and each family marks one default member. + * Legacy `MODEL_*` rows carry no #30. + * * Disabled semantics: TRUE means "this UID exists in the catalog but the * caller's account/tier cannot run inference against it." BYOK models * surface as `disabled: false` so users with their own provider keys still @@ -91,6 +101,73 @@ export interface ModelCatalogEntry { * (see src/providers/antigravity-models.ts). */ supportsImages?: boolean; + /** ModelInfo #13: the model's own output-token ceiling. */ + maxOutputTokens?: number; + /** ModelInfo #6.15. */ + supportsThinking?: boolean; + /** ModelInfo #23, e.g. `swe-1.7` for both `swe-1-7` and `swe-1-7-medium`. */ + familyUid?: string; + /** ModelFamilyMetadata #1, e.g. `SWE-1.7`. */ + familyLabel?: string; + /** + * ModelFamilyMetadata #2: this row's position on each of its family's axes, + * keyed by axis name (`Effort`, `Reasoning Effort`, `Fast Mode`, `1M Context`, + * `Thinking`, ...). Toggle axes such as `Fast Mode` carry only an order. + */ + familyAxes?: Record; + /** #31 (or #30.3): the member the family selects when nothing is asked for. */ + isFamilyDefault?: boolean; +} + +export interface DevinFamilyAxisValue { + order: number; + name?: string; +} + +/** + * Rows whose catalog entry claims image support that the model does not have. + * Live 2026-09-27: `swe-1-6` answered "NOIMAGE" to a solid red PNG while + * `swe-1-6-fast` and `swe-2-medium` named the colour, so the image is dropped + * server-side without an error. Marking it text-only lets the vision fallback + * describe the image instead of the model silently never seeing it. + */ +const IMAGE_BLIND_UIDS = new Set(['swe-1-6']); + +function parseFamilyMetadata(buf: Buffer): { label?: string; axes: Record; isDefault: boolean } { + let label: string | undefined; + let isDefault = false; + const axes: Record = {}; + for (const f of iterFields(buf)) { + if (f.num === 1 && f.wire === 2 && Buffer.isBuffer(f.value)) label = f.value.toString('utf8'); + else if (f.num === 3 && f.wire === 0) isDefault = f.value === 1n; + else if (f.num === 2 && f.wire === 2 && Buffer.isBuffer(f.value)) { + let key = ''; + const value: DevinFamilyAxisValue = { order: 0 }; + for (const e of iterFields(f.value)) { + if (e.num === 1 && e.wire === 2 && Buffer.isBuffer(e.value)) key = e.value.toString('utf8'); + else if (e.num === 2 && e.wire === 2 && Buffer.isBuffer(e.value)) { + for (const v of iterFields(e.value)) { + if (v.num === 1 && v.wire === 0) value.order = Number(v.value); + else if (v.num === 2 && v.wire === 2 && Buffer.isBuffer(v.value)) value.name = v.value.toString('utf8'); + } + } + } + if (key) axes[key] = value; + } + } + return { ...(label ? { label } : {}), axes, isDefault }; +} + +function parseModelInfo(buf: Buffer): { maxOutputTokens?: number; familyUid?: string; supportsThinking?: boolean } { + const out: { maxOutputTokens?: number; familyUid?: string; supportsThinking?: boolean } = {}; + for (const f of iterFields(buf)) { + if (f.num === 13 && typeof f.value === 'bigint' && f.value > 0n) out.maxOutputTokens = Number(f.value); + else if (f.num === 23 && f.wire === 2 && Buffer.isBuffer(f.value) && f.value.length > 0) out.familyUid = f.value.toString('utf8'); + else if (f.num === 6 && f.wire === 2 && Buffer.isBuffer(f.value)) { + for (const g of iterFields(f.value)) if (g.num === 15 && g.wire === 0) out.supportsThinking = g.value === 1n; + } + } + return out; } export interface CacheEntry { @@ -127,6 +204,9 @@ export function parseCatalogBuffer(buf: Buffer, apiKey: string, host: string): C let disabled = false; let contextWindow = 0; let supportsImages: boolean | undefined; + let info: ReturnType = {}; + let family: ReturnType | undefined; + let isFamilyDefault = false; for (const sf of iterFields(f.value as Buffer)) { if (sf.num === 1 && sf.wire === 2 && Buffer.isBuffer(sf.value)) { label = (sf.value as Buffer).toString('utf8'); @@ -145,8 +225,15 @@ export function parseCatalogBuffer(buf: Buffer, apiKey: string, host: string): C contextWindow = Number(sf.value); } else if (sf.num === 22 && sf.wire === 2 && Buffer.isBuffer(sf.value)) { modelUid = (sf.value as Buffer).toString('utf8'); + } else if (sf.num === 23 && sf.wire === 2 && Buffer.isBuffer(sf.value)) { + info = parseModelInfo(sf.value); + } else if (sf.num === 30 && sf.wire === 2 && Buffer.isBuffer(sf.value)) { + family = parseFamilyMetadata(sf.value); + } else if (sf.num === 31 && sf.wire === 0) { + isFamilyDefault = sf.value === 1n; } } + if (IMAGE_BLIND_UIDS.has(modelUid)) supportsImages = false; if (modelUid.length > 0) { byUid.set(modelUid, { modelUid, @@ -154,6 +241,10 @@ export function parseCatalogBuffer(buf: Buffer, apiKey: string, host: string): C disabled, ...(contextWindow > 0 ? { contextWindow } : {}), ...(supportsImages !== undefined ? { supportsImages } : {}), + ...info, + ...(family?.label ? { familyLabel: family.label } : {}), + ...(family ? { familyAxes: family.axes } : {}), + ...(isFamilyDefault || family?.isDefault ? { isFamilyDefault: true } : {}), }); } } diff --git a/src/adapters/devin/cloud-direct/chat.ts b/src/adapters/devin/cloud-direct/chat.ts index 30a520d3e2a..1103ac9db99 100644 --- a/src/adapters/devin/cloud-direct/chat.ts +++ b/src/adapters/devin/cloud-direct/chat.ts @@ -332,10 +332,15 @@ function collapseSystemIntoUser(messages: ChatHistoryItem[]): ChatHistoryItem[] * CompletionConfiguration — mirrors the LS-shipped defaults, lets the caller * override the obvious knobs. */ -/** Output cap when the caller named none. */ +/** Output cap when neither the caller nor the catalog named one. */ const DEFAULT_MAX_OUTPUT_TOKENS = 8192; -/** Context window when the caller named none. */ -const DEFAULT_CONTEXT_WINDOW = 128_000; +/** + * CompletionConfiguration #3 is `max_newlines`, not a token count and not an + * input ceiling: live, a value of 5 did not truncate a 25-line answer. It is + * still sent, at the value every turn has carried, so the request shape the + * service accepts does not change. + */ +const MAX_NEWLINES = 128_000; /** * Cognition rejects a temperature of exactly 0 with the same opaque internal @@ -353,7 +358,6 @@ function safeTemperature(value: number | undefined): number { function encodeCompletionConfiguration(opts: { maxOutputTokens?: number; - maxInputTokens?: number; temperature?: number; topK?: number; topP?: number; @@ -365,15 +369,15 @@ function encodeCompletionConfiguration(opts: { }; // Tag map, verified by building the same turn with a working client and // diffing the encoded messages field by field: #2 is the OUTPUT cap and #3 is - // the context window. This layout had those two swapped, so a caller asking - // for 32 output tokens put 32 into the context-window field and the request - // came back as an opaque "an internal error occurred" — for every turn, on - // every account, which is why free and paid failed identically. #6 and #11 - // are not part of the message the service accepts. + // max_newlines. This layout once had those two swapped, so a caller's output + // cap landed in #3 and a large value in #2, and the request came back as an + // opaque "an internal error occurred" — for every turn, on every account, + // which is why free and paid failed identically. #6 and #11 are not part of + // the message the service accepts. return Buffer.concat([ encodeVarintField(1, 1), encodeVarintField(2, opts.maxOutputTokens ?? DEFAULT_MAX_OUTPUT_TOKENS), - encodeVarintField(3, opts.maxInputTokens ?? DEFAULT_CONTEXT_WINDOW), + encodeVarintField(3, MAX_NEWLINES), enc64(5, safeTemperature(opts.temperature)), encodeVarintField(7, opts.topK ?? 40), enc64(8, opts.topP ?? 1.0), @@ -543,7 +547,6 @@ interface BuildArgs { requestType?: number; completionOpts?: { maxOutputTokens?: number; - maxInputTokens?: number; temperature?: number; topK?: number; topP?: number; diff --git a/src/adapters/devin/live-models.ts b/src/adapters/devin/live-models.ts index f9eaca88954..e90a3441491 100644 --- a/src/adapters/devin/live-models.ts +++ b/src/adapters/devin/live-models.ts @@ -2,12 +2,12 @@ * Live Devin / Cognition model discovery via GetCascadeModelConfigs. * * The live catalog is the source of truth for the model roster. The endpoint - * returns effort-suffixed variants (e.g. `gpt-5-6-sol-high`); we collapse those - * to base ids so the picker stays clean and the adapter appends the effort - * suffix at request time. `DEVIN_STATIC_MODELS` is only a degraded-mode + * returns one row per variant (e.g. `gpt-5-6-sol-high`), grouped into families; + * the picker shows one id per family and the adapter picks the family member + * for the requested effort at request time. `DEVIN_STATIC_MODELS` is only a degraded-mode * fallback for when there is no API key or discovery fails. */ -import { getCachedCatalog, type ModelCatalogEntry } from "./cloud-direct"; +import { getCachedCatalog, type CacheEntry, type ModelCatalogEntry } from "./cloud-direct"; const DEFAULT_HOST = "https://server.codeium.com"; @@ -129,6 +129,139 @@ export function sortDevinRungs(rungs: Iterable): string[] { return [...new Set(rungs)].sort((a, b) => RUNG_ORDER.indexOf(a) - RUNG_ORDER.indexOf(b)); } +/** + * Effort rungs a catalog family axis can name, in ladder order. Wider than + * RUNG_ORDER because the catalog also spells `Minimal` (Gemini Flash) and + * `No Thinking` (GLM), which resolution has to place even though the Codex + * ladder does not offer them. + */ +const FAMILY_EFFORT_LADDER = ["none", "minimal", "low", "medium", "high", "xhigh", "max"]; +const EFFORT_AXES = ["Effort", "Reasoning Effort"]; +const FAST_AXIS = "Fast Mode"; +const LONG_CONTEXT_AXIS = "1M Context"; +const THINKING_AXIS = "Thinking"; + +/** The picker id of a catalog family: its uid with dotted versions hyphenated, as the row uids spell it. */ +export function devinFamilyBaseId(familyUid: string): string { + return familyUid.replace(/\./g, "-"); +} + +function isEffortAxis(axis: string): boolean { + return EFFORT_AXES.includes(axis); +} + +/** The effort rung a family member sits on, read from its axis value name rather than its uid. */ +export function devinFamilyEffortOf(entry: ModelCatalogEntry): string | undefined { + const axes = entry.familyAxes; + if (!axes) return undefined; + for (const axis of EFFORT_AXES) { + const name = axes[axis]?.name?.toLowerCase(); + if (!name) continue; + const rung = name === "no thinking" ? "none" : name; + if (FAMILY_EFFORT_LADDER.includes(rung)) return rung; + } + return undefined; +} + +const familyCache = new WeakMap>(); + +/** + * Group catalog rows by family, keyed by the family's picker id. Only rows + * that carry family metadata take part; legacy `MODEL_*` rows are internal + * enum names and never form a picker family. + */ +export function devinFamiliesOf(catalog: CacheEntry): Map { + let families = familyCache.get(catalog); + if (families) return families; + families = new Map(); + for (const entry of catalog.byUid.values()) { + if (!entry.familyUid || !entry.familyAxes || entry.modelUid.startsWith("MODEL_")) continue; + const base = devinFamilyBaseId(entry.familyUid); + const members = families.get(base); + if (members) members.push(entry); + else families.set(base, [entry]); + } + familyCache.set(catalog, families); + return families; +} + +/** What a caller asked for, independent of how the catalog spells it. */ +export interface DevinVariantRequest { + effort?: string; + fast?: boolean; + longContext?: boolean; +} + +function lexicallyLess(a: number[], b: number[]): boolean { + for (let i = 0; i < a.length; i++) if (a[i] !== b[i]) return a[i]! < b[i]!; + return false; +} + +function effortIndex(rung: string | undefined): number { + return rung === undefined ? -1 : FAMILY_EFFORT_LADDER.indexOf(rung); +} + +/** + * The member closest to `targets` on the non-effort axes, then the lowest rung + * at or above `effort`, falling back to the highest rung below it. A missing + * rung never quietly turns reasoning down: `high` on SWE-1.7 (Medium, Max) + * selects Max, `xhigh` on SWE-2 selects Max, and `medium` on Kimi K3 (Low, + * High, Max) selects High. + */ +function closestMember( + members: ModelCatalogEntry[], + targets: Record, + effort: string | undefined, +): ModelCatalogEntry | undefined { + const want = effortIndex(effort); + let best: ModelCatalogEntry | undefined; + let bestScore: number[] | undefined; + for (const member of members) { + const axes = member.familyAxes ?? {}; + const mismatches = Object.entries(targets).filter(([axis, order]) => (axes[axis]?.order ?? 0) !== order).length; + const have = effortIndex(devinFamilyEffortOf(member)); + const distance = want < 0 ? 0 + : have < 0 ? 2 * FAMILY_EFFORT_LADDER.length + : have >= want ? have - want : FAMILY_EFFORT_LADDER.length + (want - have); + const score = [mismatches, distance, -have]; + if (!bestScore || lexicallyLess(score, bestScore)) { + best = member; + bestScore = score; + } + } + return best; +} + +/** + * Pick the family member for a request. Axes the caller did not ask about + * (`Fast Mode`, `1M Context`, `Thinking`, `Prompt Cache Retention`, ...) stay + * where the anchor has them; the anchor is the member the caller named, else + * the family's catalog default, else the neutral member (every toggle off, + * effort nearest Medium) for the few families that mark no default. Disabled + * members are passed over while any enabled one remains, so the pre-flight + * only reports a tier refusal when the whole family is refused. + */ +export function selectDevinFamilyMember( + members: ModelCatalogEntry[], + request: DevinVariantRequest, + anchor?: ModelCatalogEntry, +): ModelCatalogEntry | undefined { + const axisNames = new Set(members.flatMap((m) => Object.keys(m.familyAxes ?? {})).filter((a) => !isEffortAxis(a))); + const base = anchor + ?? members.find((m) => m.isFamilyDefault) + ?? closestMember(members, Object.fromEntries([...axisNames].map((a) => [a, 0])), "medium"); + if (!base) return undefined; + const targets: Record = {}; + for (const axis of axisNames) targets[axis] = base.familyAxes?.[axis]?.order ?? 0; + if (request.fast && axisNames.has(FAST_AXIS)) targets[FAST_AXIS] = 1; + if (request.longContext && axisNames.has(LONG_CONTEXT_AXIS)) targets[LONG_CONTEXT_AXIS] = 1; + // `none` means no reasoning. Families like Claude Sonnet 4.6 express that as + // Thinking off rather than as an effort rung. + if (request.effort === "none" && axisNames.has(THINKING_AXIS)) targets[THINKING_AXIS] = 0; + const enabled = members.filter((m) => !m.disabled); + return closestMember(enabled.length > 0 ? enabled : members, targets, request.effort ?? devinFamilyEffortOf(base)); +} + /** * Degraded-mode ladders, used only before the account catalog is readable. * @@ -158,14 +291,18 @@ export type DevinUsableModelsResult = models: string[]; contextWindows: Record; efforts: Record; + /** Per base, the effort of the family's catalog default member, when it is on the ladder. */ + defaultEfforts: Record; inputModalities: Record; } | { ok: false; error: "auth" | "http" | "empty" | "unknown"; detail?: string }; /** * Fetch the live model roster from Cognition's `GetCascadeModelConfigs` and - * collapse effort-suffixed variants to base ids. The returned list is the - * authoritative model roster for the signed-in account. + * collapse variants to base ids. Rows that carry family metadata collapse to + * their family and read their effort from the family's effort axis; only rows + * without it (older catalogs) fall back to reading suffixes off the uid. The + * returned list is the authoritative model roster for the signed-in account. */ export async function fetchDevinUsableModels(opts: { apiKey: string; @@ -180,6 +317,7 @@ export async function fetchDevinUsableModels(opts: { const contextWindows: Record = {}; // Effort rungs per base, recovered from the suffixes the collapse strips. const rungs = new Map>(); + const defaultRungs = new Map(); // supportsImages votes per base; only rows that asserted field #5 vote. const imageVotes = new Map(); for (const entry of catalog.byUid.values()) { @@ -187,9 +325,14 @@ export async function fetchDevinUsableModels(opts: { // Skip internal enum constants (e.g. MODEL_GPT_5_2_LOW, MODEL_PRIVATE_*). // Real chat model UIDs are lowercase dashed strings (swe-1-7, gpt-5-6-sol). if (entry.modelUid.startsWith("MODEL_")) continue; - const base = collapseDevinModelUid(entry.modelUid); + const family = entry.familyUid && entry.familyAxes ? devinFamilyBaseId(entry.familyUid) : undefined; + const base = family ?? collapseDevinModelUid(entry.modelUid); bases.add(base); - const found = devinReasoningRungsOf(entry.modelUid); + const familyRung = family ? devinFamilyEffortOf(entry) : undefined; + const found = family + ? (familyRung && REASONING_RUNG_TOKENS.has(familyRung) ? [familyRung] : []) + : devinReasoningRungsOf(entry.modelUid); + if (family && entry.isFamilyDefault && familyRung) defaultRungs.set(base, familyRung); if (found.length > 0) { let set = rungs.get(base); if (!set) { set = new Set(); rungs.set(base, set); } @@ -222,13 +365,18 @@ export async function fetchDevinUsableModels(opts: { // would draw a picker whose only option is the value already in effect. if (set.size > 1) efforts[base] = sortDevinRungs(set); } + const defaultEfforts: Record = {}; + for (const [base, rung] of defaultRungs) { + if (efforts[base]?.includes(rung)) defaultEfforts[base] = rung; + } // supportsImages arrives tri-state per catalog row, so the collapse votes: // a row that never asserted field #5 abstains, which keeps an unsuffixed // unknown row from poisoning a base whose effort variants were measured // image-capable. Unanimous measured rows advertise; measured disagreement // advertises nothing, because a single measured false is not outvoted by - // its siblings. One accepted mismatch: resolveWireModelUid prefers the - // plain UID when the catalog lists it, so a base advertised + // its siblings. One accepted mismatch, only for rows without family + // metadata: resolveWireModelUid prefers the plain UID when the catalog + // lists it, so a base advertised // ["text","image"] on variant evidence can still route a no-effort request // to a plain row that never asserted the field. const inputModalities: Record = {}; @@ -236,7 +384,7 @@ export async function fetchDevinUsableModels(opts: { if (votes.sawTrue && votes.sawFalse) continue; inputModalities[base] = votes.sawTrue ? ["text", "image"] : ["text"]; } - return { ok: true, models: [...bases].sort(), contextWindows, efforts, inputModalities }; + return { ok: true, models: [...bases].sort(), contextWindows, efforts, defaultEfforts, inputModalities }; } catch (error) { const message = error instanceof Error ? error.message : String(error); if (/unauth|401|invalid token|login/i.test(message)) return { ok: false, error: "auth", detail: message }; diff --git a/src/codex/catalog/provider-models.ts b/src/codex/catalog/provider-models.ts index 3c2185b83ad..a4e8e8fdd19 100644 --- a/src/codex/catalog/provider-models.ts +++ b/src/codex/catalog/provider-models.ts @@ -424,6 +424,8 @@ export async function fetchProviderModelsWithAuth( // away, and every client that keys an effort control off this field — // the Pi-shaped exports — renders no control at all. ...(liveResult.efforts[id]?.length ? { reasoningEfforts: liveResult.efforts[id] } : {}), + // The family's catalog default member, so an unset picker lands where Cognition does. + ...(liveResult.defaultEfforts[id] ? { defaultReasoningEffort: liveResult.defaultEfforts[id] } : {}), // The account catalog's per-base supportsImages vote collapses to one // modalities value. It spreads before the hints so exact // modelCapabilities declarations, the legacy modelInputModalities diff --git a/structure/catalog.md b/structure/catalog.md index 0c79ecbc20c..4aea05f5816 100644 --- a/structure/catalog.md +++ b/structure/catalog.md @@ -226,7 +226,7 @@ in-flight publication; unrelated unscoped rows require no credential lookup. A Devin live row spreads its measured `inputModalities` before `catalogHintsFromProviderConfig`, so exact `modelCapabilities` declarations, the legacy `modelInputModalities` record and the vision-sidecar rewrite keep precedence and the live -value survives only when none of them applies. +value survives only when none of them applies. Devin live rows collapse by catalog family (so `swe-1-6-fast` stays its own row), read their ladder from the family effort axis, and carry the family default member's effort as `defaultReasoningEffort`; `swe-1-6` is marked text-only because it drops images without an error. For `liveModels: false`, a static provider publishes the ordered union of `models` and `retainModels`. When `models` is absent or empty, its configured `defaultModel` seeds that union before retained ids; a nonempty explicit list does not import a different default. diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index ad719ac37ec..c90fcbed666 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -127,7 +127,7 @@ rewrite rules and the routed-id settlement. | `src/adapters/declaration-carrier.ts`, `src/adapters/input-media-guard.ts` | Default-deny allowlists for constraints the normalized request carries but a wire may not be able to express: `tools[*].allowed_callers`, which fences a tool off from callers, and inline document bytes. Both are refused with a 400 at the single guard every registered adapter passes through, rather than left to each adapter, because an adapter that never learned about the carrier rebuilds without it and answers normally. `allowed_callers` reaches the `anthropic` wire; document bytes reach `anthropic`, `openai-chat` and `google`; the `openai-responses` wire is exempt from the whole guard because it forwards the original body. Adding an `AdapterWire` member makes the omission visible in these lists instead of at a customer's upstream. The unrestricted `["direct"]` caller default is not a restriction. | | `src/adapters/azure.ts` | Azure OpenAI bridge. | | `src/adapters/cursor.ts`, `src/adapters/cursor/` | Cursor protobuf transport: discovery, request builder, event decoding, MCP, thread continuity, native-exec policy. | -| `src/adapters/devin.ts`, `src/adapters/devin/cloud-direct/` | Devin runTurn transport over Cognition Connect-RPC. `GetChatMessage` uses the Responses provider executor and shared physical-send budget; catalog, JWT, and `src/web-search/devin-executor.ts` native search support RPCs remain outside inference-send accounting. Provider-stated pre-output 429 reset delays are surfaced immediately by default, releasing shared active-turn capacity. A positive `OPENCODEX_DEVIN_STATED_RESET_WAIT_MS` explicitly enables bounded waiting and up to two replays on standalone turns, which hold that capacity until completion or cancellation. Combo children bypass that wait and surface a pre-output 429 so the next target can run. During an opted-in standalone wait, safe heartbeats commit the response preflight and keep the stream's stall watchdog fed. Invalid values fail closed to the immediate-refusal behavior. A recorded tenant host is used only for the stored account whose credential owns the transmitted key, searched in the configured provider id and then its deprecated alias; a configured, forwarded, or unmatched key uses the configured base URL or the US default. Native search previews the current route by effective adapter without mutating combo selection state, pins one admitted active-account snapshot for the request, and calls `GetWebSearchResults`, so it starts no CLI or second model. | +| `src/adapters/devin.ts`, `src/adapters/devin/cloud-direct/` | Devin runTurn transport over Cognition Connect-RPC. `GetChatMessage` uses the Responses provider executor and shared physical-send budget; catalog, JWT, and `src/web-search/devin-executor.ts` native search support RPCs remain outside inference-send accounting. Provider-stated pre-output 429 reset delays are surfaced immediately by default, releasing shared active-turn capacity. A positive `OPENCODEX_DEVIN_STATED_RESET_WAIT_MS` explicitly enables bounded waiting and up to two replays on standalone turns, which hold that capacity until completion or cancellation. Combo children bypass that wait and surface a pre-output 429 so the next target can run. During an opted-in standalone wait, safe heartbeats commit the response preflight and keep the stream's stall watchdog fed. Invalid values fail closed to the immediate-refusal behavior. A recorded tenant host is used only for the stored account whose credential owns the transmitted key, searched in the configured provider id and then its deprecated alias; a configured, forwarded, or unmatched key uses the configured base URL or the US default. Native search previews the current route by effective adapter without mutating combo selection state, pins one admitted active-account snapshot for the request, and calls `GetWebSearchResults`, so it starts no CLI or second model. The wire model UID comes from the catalog's family metadata (`ClientModelConfig` #23/#30/#31): a family id with no effort selects the family's default member, an effort moves only the effort axis, to the lowest rung at or above it (else the highest below) while Fast Mode, 1M Context and the other axes stay at the anchor's values unless the caller asked for `fast` or a `1m` value, and resolution never leaves the family. Rows without family metadata fall back to suffix resolution, whose variant scan matches the collapsed base rather than a string prefix. With no caller or configured output cap, the selected row's catalog `maxOutputTokens` fills CompletionConfiguration #2; #3 is `max_newlines` and is sent at a fixed value, never a context window. | | `src/adapters/kiro.ts` and `src/adapters/kiro/` | Kiro event/tool/thinking/truncation/retry handling, including an egress-aware completion fallback, fixed public HTTP 5xx text, and closed-set status/code diagnostics. The original path is a facade over leaves for wire identity, reasoning, conversation state, token estimation, payload assembly, streaming, and the adapter. | | `src/adapters/mimo-free.ts` | Mimo Free transport (client identity + JWT). Concurrent requests share one JWT bootstrap bound only to its timeout; each request stops waiting on its own abort without cancelling the others. | | `src/adapters/command-code.ts`, `src/adapters/command-code-tool-text.ts`, `src/adapters/command-code-restored-schema.ts` | Command Code OAuth NDJSON translation. For every `xiaomi/mimo-` model, text, native calls, reasoning, and terminal decisions share one byte-bounded queue with linear queue visits. Markup is deduplicated against matching native calls; text-only restoration requires one contiguous text run, a clean finish, a declared tool, and arguments validated against supported schema constraints. A parameter-free (freeform) block may omit `` but must end with ``; parameter blocks keep the canonical close. Markup appended after prose in the same delta is split off at the marker and held like a block that opens with ``; a marker split across deltas after prose is still released as text. Native, reasoning, and other intervening events interrupt a still-probing block but leave a held block held in arrival order, and the queued byte bound still flushes an unresolved envelope as text. An envelope the strict parser rejects but that opens with ``, closes with ``, and names a declared function is dropped when a native call for that same function arrives and on a clean finish; markup that parses but fits no supported schema is still released as text. Regex patterns, other unsupported constraints, and abnormal finishes fail closed. `tests/providers/command-code-tool-text-prose-split.test.ts` covers the split, the interleaved-event hold, and both drop paths. | diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 01768b541df..b41757b4f41 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -750,6 +750,7 @@ "devin-adapter.test.ts": "providers", "devin-cli-authmode-migration.test.ts": "providers", "devin-effort-ladder.test.ts": "providers", + "devin-family-resolution.test.ts": "providers", "devin-hardening.test.ts": "providers", "devin-image-passthrough.test.ts": "providers", "devin-live-models.test.ts": "providers", diff --git a/tests/providers/devin-adapter.test.ts b/tests/providers/devin-adapter.test.ts index 46554e60839..76e79208a5f 100644 --- a/tests/providers/devin-adapter.test.ts +++ b/tests/providers/devin-adapter.test.ts @@ -379,12 +379,14 @@ describe("devin adapter", () => { }); describe("SWE-2 wire effort selection", () => { + // Every case here passes a null catalog: this is the degraded path. With a + // catalog, family metadata decides (devin-family-resolution.test.ts). // Cognition spells SWE-2 effort as the model id, so an explicit effort has to // beat a suffix the picker already chose. Before this, swe-2-high asked for at // medium stayed high and the caller was silently ignored. test.each(["medium", "high", "max"])("an explicit %s effort overrides every SWE-2 variant", async (effort) => { for (const model of ["swe-2", "swe-2-medium", "swe-2-high", "swe-2-max", "swe-2.high"]) { - expect(await resolveWireModelUidForTests(model, "unused", "unused", effort)).toBe(`swe-2-${effort}`); + expect(await resolveWireModelUidForTests(model, "unused", "unused", effort, null)).toBe(`swe-2-${effort}`); } }); @@ -392,23 +394,23 @@ describe("SWE-2 wire effort selection", () => { ["none", "medium"], ["off", "medium"], ["minimal", "medium"], ["low", "medium"], ["xhigh", "max"], ["ultra", "max"], ])("maps %s to the supported SWE-2 %s lane", async (effort, expected) => { - expect(await resolveWireModelUidForTests("swe-2-high", "unused", "unused", effort)).toBe(`swe-2-${expected}`); + expect(await resolveWireModelUidForTests("swe-2-high", "unused", "unused", effort, null)).toBe(`swe-2-${expected}`); }); // Case is normalised, which the source contribution did not do: a caller that // sends HIGH means the same lane as high. test("effort matching is case-insensitive", async () => { - expect(await resolveWireModelUidForTests("swe-2-medium", "unused", "unused", "HIGH")).toBe("swe-2-high"); + expect(await resolveWireModelUidForTests("swe-2-medium", "unused", "unused", "HIGH", null)).toBe("swe-2-high"); }); test("omitted or unknown effort preserves an explicit variant", async () => { - expect(await resolveWireModelUidForTests("swe-2-high", "unused", "unused")).toBe("swe-2-high"); - expect(await resolveWireModelUidForTests("swe-2-max", "unused", "unused", "future-effort")).toBe("swe-2-max"); + expect(await resolveWireModelUidForTests("swe-2-high", "unused", "unused", undefined, null)).toBe("swe-2-high"); + expect(await resolveWireModelUidForTests("swe-2-max", "unused", "unused", "future-effort", null)).toBe("swe-2-max"); }); - test("other model families keep their existing suffix precedence", async () => { + test("without a catalog, other model families keep their suffix precedence", async () => { for (const model of ["claude-opus-5-medium", "gpt-5-6-sol-high", "swe-1-7-high", "swe-20-high"]) { - expect(await resolveWireModelUidForTests(model, "unused", "unused", "max")).toBe(model); + expect(await resolveWireModelUidForTests(model, "unused", "unused", "max", null)).toBe(model); } }); }); @@ -419,24 +421,24 @@ describe("effort suffix detection and caller effort values are different sets", // the exact shape Cognition answers with an opaque permission_denied. test("a UID carrying the priority tier is recognised as already suffixed", async () => { for (const uid of ["gpt-5-6-sol-priority", "gpt-5-6-sol-medium-priority"]) { - expect(await resolveWireModelUidForTests(uid, "unused", "unused", "high")).toBe(uid); + expect(await resolveWireModelUidForTests(uid, "unused", "unused", "high", null)).toBe(uid); } }); test("detection handles a compound suffix, which a last-token test could not", async () => { - expect(await resolveWireModelUidForTests("gpt-5-6-sol-medium-priority", "unused", "unused")).toBe( + expect(await resolveWireModelUidForTests("gpt-5-6-sol-medium-priority", "unused", "unused", undefined, null)).toBe( "gpt-5-6-sol-medium-priority", ); }); test("a bare model still receives the caller effort", async () => { - expect(await resolveWireModelUidForTests("gpt-5-6-sol", "unused", "unused", "high")).toBe("gpt-5-6-sol-high"); + expect(await resolveWireModelUidForTests("gpt-5-6-sol", "unused", "unused", "high", null)).toBe("gpt-5-6-sol-high"); }); // `priority` is a service tier, not something a caller asks for as effort. // Sharing one set between detection and caller validity would admit it. test("priority is not accepted as a caller reasoning effort", async () => { - expect(await resolveWireModelUidForTests("gpt-5-6-sol", "unused", "unused", "priority")).toBe( + expect(await resolveWireModelUidForTests("gpt-5-6-sol", "unused", "unused", "priority", null)).toBe( "gpt-5-6-sol-medium", ); }); @@ -444,7 +446,7 @@ describe("effort suffix detection and caller effort values are different sets", // These never appear as a trailing token, so they are meaningless to detection, // but a caller can still name them and they must survive. test.each(["max-1m", "none-1m", "1m", "fast"])("the compound caller value %p is preserved", async (effort) => { - expect(await resolveWireModelUidForTests("gpt-5-6-sol", "unused", "unused", effort)).toBe( + expect(await resolveWireModelUidForTests("gpt-5-6-sol", "unused", "unused", effort, null)).toBe( `gpt-5-6-sol-${effort}`, ); }); @@ -452,7 +454,7 @@ describe("effort suffix detection and caller effort values are different sets", test("a model name is never mistaken for a suffix", async () => { // Greedy collapse must not eat part of a real model name. for (const uid of ["claude-opus-5", "swe-1-7", "glm-5-3"]) { - expect(await resolveWireModelUidForTests(uid, "unused", "unused", "high")).toBe(`${uid}-high`); + expect(await resolveWireModelUidForTests(uid, "unused", "unused", "high", null)).toBe(`${uid}-high`); } }); }); diff --git a/tests/providers/devin-family-resolution.test.ts b/tests/providers/devin-family-resolution.test.ts new file mode 100644 index 00000000000..114aeab4001 --- /dev/null +++ b/tests/providers/devin-family-resolution.test.ts @@ -0,0 +1,281 @@ +/** + * Devin model resolution over the catalog's family metadata + * (ClientModelConfig #23 ModelInfo, #30 ModelFamilyMetadata, #31 default flag). + * + * Row shapes mirror a live GetCascadeModelConfigs response captured 2026-09-27: + * bare `swe-1-7` is the Max row while `swe-1-7-medium` is the family default, + * SWE-2 has no bare row, GLM-5.2's bare row is its default, Fast Mode is a + * `-fast` suffix on Claude and `-priority` on GPT, and three families share a + * string prefix with a different family. + */ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { resolveWireModelUidForTests } from "../../src/adapters/devin"; +import { fetchDevinUsableModels } from "../../src/adapters/devin/live-models"; +import { parseCatalogBuffer, setCachedCatalogForTests, type CacheEntry } from "../../src/adapters/devin/cloud-direct/catalog"; +import { encodeMessage, encodeString, encodeVarintField } from "../../src/adapters/devin/cloud-direct/wire"; + +const HOST = "https://server.codeium.com"; +const KEY = "devin-family-resolution-test-key"; + +interface Row { + uid: string; + family?: string; + axes?: Array<[key: string, order: number, name?: string]>; + isDefault?: boolean; + maxOut?: number; + images?: boolean; + disabled?: boolean; + thinking?: boolean; +} + +function row(r: Row): Buffer { + const info = Buffer.concat([ + ...(r.thinking !== undefined ? [encodeMessage(6, encodeVarintField(15, r.thinking ? 1 : 0))] : []), + ...(r.maxOut !== undefined ? [encodeVarintField(13, r.maxOut)] : []), + ...(r.family ? [encodeString(23, r.family)] : []), + ]); + const meta = r.axes ? Buffer.concat([ + encodeString(1, r.family ?? ""), + ...r.axes.map(([key, order, name]) => encodeMessage(2, Buffer.concat([ + encodeString(1, key), + encodeMessage(2, Buffer.concat([ + ...(order ? [encodeVarintField(1, order)] : []), + ...(name ? [encodeString(2, name)] : []), + ])), + ]))), + ...(r.isDefault ? [encodeVarintField(3, 1)] : []), + ]) : undefined; + return encodeMessage(1, Buffer.concat([ + encodeString(1, r.uid), + ...(r.disabled ? [encodeVarintField(4, 1)] : []), + ...(r.images !== undefined ? [encodeVarintField(5, r.images ? 1 : 0)] : []), + encodeString(22, r.uid), + ...(info.length > 0 ? [encodeMessage(23, info)] : []), + ...(meta ? [encodeMessage(30, meta)] : []), + ...(r.isDefault ? [encodeVarintField(31, 1)] : []), + ])); +} + +const EFFORT5 = ["Low", "Medium", "High", "XHigh", "Max"] as const; + +function claudeFamily(family: string, prefix: string): Row[] { + const rows: Row[] = []; + for (const fast of [0, 1]) { + EFFORT5.forEach((name, order) => rows.push({ + uid: `${prefix}-${name.toLowerCase()}${fast ? "-fast" : ""}`, + family, + axes: [["Effort", order, name], ["Thinking", 1], ["Fast Mode", fast], ["1M Context", 0]], + isDefault: !fast && name === "Medium", + maxOut: 128_000, + images: true, + thinking: true, + })); + } + return rows; +} + +function gptFamily(family: string, prefix: string, withDefault = true): Row[] { + const rows: Row[] = []; + const names = ["None", "Low", "Medium", "High", "XHigh", "Max"]; + for (const fast of [0, 1]) { + names.forEach((name, order) => rows.push({ + uid: `${prefix}-${name.toLowerCase()}${fast ? "-priority" : ""}`, + family, + axes: [["Reasoning Effort", order, name], ["Fast Mode", fast], ["Prompt Cache Retention", 1, "24h"]], + isDefault: withDefault && !fast && name === "Medium", + maxOut: 128_000, + })); + } + return rows; +} + +const LIVE_ROWS: Row[] = [ + { uid: "swe-1-7", family: "swe-1.7", axes: [["Reasoning Effort", 5, "Max"]], maxOut: 128_000, images: true }, + { uid: "swe-1-7-medium", family: "swe-1.7", axes: [["Reasoning Effort", 2, "Medium"]], isDefault: true, maxOut: 128_000, images: true }, + { uid: "swe-2-medium", family: "swe-2", axes: [["Reasoning Effort", 0, "Medium"]], maxOut: 128_000, images: true }, + { uid: "swe-2-high", family: "swe-2", axes: [["Reasoning Effort", 1, "High"]], isDefault: true, maxOut: 128_000, images: true }, + { uid: "swe-2-max", family: "swe-2", axes: [["Reasoning Effort", 2, "Max"]], maxOut: 128_000, images: true }, + ...claudeFamily("claude-opus-5", "claude-opus-5"), + ...claudeFamily("claude-opus-5-5", "claude-opus-5-5"), + ...gptFamily("gpt-6-sol", "gpt-6-sol"), + // gpt-5.4 marks no default member; gpt-5.4-mini shares its string prefix. + ...gptFamily("gpt-5.4", "gpt-5-4", false), + { uid: "gpt-5-4-mini-low", family: "gpt-5.4-mini", axes: [["Reasoning Effort", 0, "Low"]] }, + { uid: "glm-5-2", family: "glm-5.2", axes: [["Effort", 1, "High"], ["1M Context", 0]], isDefault: true }, + { uid: "glm-5-2-max", family: "glm-5.2", axes: [["Effort", 2, "Max"], ["1M Context", 0]] }, + { uid: "glm-5-2-none", family: "glm-5.2", axes: [["Effort", 0, "No Thinking"], ["1M Context", 0]] }, + { uid: "glm-5-2-1m", family: "glm-5.2", axes: [["Effort", 1, "High"], ["1M Context", 1]] }, + { uid: "glm-5-2-max-1m", family: "glm-5.2", axes: [["Effort", 2, "Max"], ["1M Context", 1]] }, + { uid: "glm-5-2-none-1m", family: "glm-5.2", axes: [["Effort", 0, "No Thinking"], ["1M Context", 1]] }, + { uid: "glm-5-3-high", family: "glm-5-3", axes: [["Effort", 1, "High"]], isDefault: true }, + { uid: "glm-5-3-flash-low", family: "glm-5-3-flash", axes: [["Effort", 0, "Low"]], isDefault: true }, + { uid: "kimi-k3-low", family: "kimi-k3", axes: [["Reasoning Effort", 0, "Low"]], maxOut: 131_072, images: true }, + { uid: "kimi-k3-high", family: "kimi-k3", axes: [["Reasoning Effort", 1, "High"]], isDefault: true, maxOut: 131_072, images: true }, + { uid: "kimi-k3-max", family: "kimi-k3", axes: [["Reasoning Effort", 2, "Max"]], maxOut: 131_072, images: true }, + { uid: "claude-sonnet-4-6", family: "claude-sonnet-4.6", axes: [["Effort", 2, "High"], ["Thinking", 0], ["Fast Mode", 0], ["1M Context", 0]] }, + { uid: "claude-sonnet-4-6-thinking", family: "claude-sonnet-4.6", axes: [["Effort", 2, "High"], ["Thinking", 1], ["Fast Mode", 0], ["1M Context", 0]], isDefault: true }, + { uid: "claude-sonnet-4-6-1m", family: "claude-sonnet-4.6", axes: [["Effort", 2, "High"], ["Thinking", 0], ["Fast Mode", 0], ["1M Context", 1]] }, + { uid: "gemini-3-5-flash-minimal", family: "gemini-3.5-flash", axes: [["Reasoning Effort", 0, "Minimal"]] }, + { uid: "gemini-3-5-flash-medium", family: "gemini-3.5-flash", axes: [["Reasoning Effort", 2, "Medium"]], isDefault: true }, + // Both assert #5 true; only swe-1-6 is measured image-blind. + { uid: "swe-1-6", family: "swe-1.6", axes: [["Speed", 0]], images: true, maxOut: 128_000 }, + { uid: "swe-1-6-fast", family: "swe-1.6-fast", axes: [["Speed", 1]], images: true, maxOut: 128_000 }, + // Legacy rows carry no family metadata. + { uid: "MODEL_PRIVATE_11", maxOut: 64_000 }, +]; + +function catalogOf(rows: Row[]): CacheEntry { + return parseCatalogBuffer(Buffer.concat(rows.map(row)), KEY, HOST); +} + +const resolve = (catalog: CacheEntry, model: string, effort?: string) => + resolveWireModelUidForTests(model, KEY, HOST, effort, catalog); + +describe("catalog family metadata parsing", () => { + const catalog = catalogOf(LIVE_ROWS); + + test("reads ModelInfo, family axes and the default flag", () => { + expect(catalog.byUid.get("swe-1-7-medium")).toMatchObject({ + familyUid: "swe-1.7", + familyLabel: "swe-1.7", + familyAxes: { "Reasoning Effort": { order: 2, name: "Medium" } }, + isFamilyDefault: true, + maxOutputTokens: 128_000, + }); + expect(catalog.byUid.get("swe-1-7")?.isFamilyDefault).toBeUndefined(); + expect(catalog.byUid.get("claude-opus-5-low-fast")).toMatchObject({ + supportsThinking: true, + familyAxes: { "Fast Mode": { order: 1 }, Thinking: { order: 1 }, Effort: { order: 0, name: "Low" } }, + }); + expect(catalog.byUid.get("MODEL_PRIVATE_11")).toMatchObject({ maxOutputTokens: 64_000 }); + expect(catalog.byUid.get("MODEL_PRIVATE_11")?.familyAxes).toBeUndefined(); + }); + + test("swe-1-6 is text-only even though its row asserts images; swe-1-6-fast is not", () => { + expect(catalog.byUid.get("swe-1-6")?.supportsImages).toBe(false); + expect(catalog.byUid.get("swe-1-6-fast")?.supportsImages).toBe(true); + }); +}); + +describe("family-based wire model resolution", () => { + const catalog = catalogOf(LIVE_ROWS); + + test.each([ + // An effort moves only the effort axis; bare swe-1-7 is Max, not the default. + ["swe-1-7", "medium", "swe-1-7-medium"], + ["swe-1-7", undefined, "swe-1-7-medium"], + ["swe-1.7", undefined, "swe-1-7-medium"], + ["swe-1-7", "max", "swe-1-7"], + // A missing rung rounds up, never down: SWE-1.7 has only Medium and Max. + ["swe-1-7", "high", "swe-1-7"], + ["swe-1-7", "low", "swe-1-7-medium"], + ["swe-1-7-high", undefined, "swe-1-7"], + ["swe-2", undefined, "swe-2-high"], + ["swe-2", "max", "swe-2-max"], + ["swe-2", "low", "swe-2-medium"], + ["swe-2", "xhigh", "swe-2-max"], + ["swe-2-high", "medium", "swe-2-medium"], + ["glm-5-2", undefined, "glm-5-2"], + ["glm-5.2", undefined, "glm-5-2"], + ["glm-5-2", "none", "glm-5-2-none"], + ["glm-5-2", "max-1m", "glm-5-2-max-1m"], + ["glm-5-2", "1m", "glm-5-2-1m"], + // Fast Mode and 1M Context stay at the default member's values unless asked. + ["claude-opus-5", undefined, "claude-opus-5-medium"], + ["claude-opus-5", "fast", "claude-opus-5-medium-fast"], + ["claude-opus-5", "HIGH", "claude-opus-5-high"], + ["claude-opus-5-high-fast", "low", "claude-opus-5-low-fast"], + ["claude-opus-5-high-fast", undefined, "claude-opus-5-high-fast"], + ["gpt-6-sol", "fast", "gpt-6-sol-medium-priority"], + ["gpt-6-sol", "none", "gpt-6-sol-none"], + ["gpt-6-sol", "priority", "gpt-6-sol-medium"], + ["gpt-6-sol-medium-priority", undefined, "gpt-6-sol-medium-priority"], + // Nearest rung, ties to the higher one; never another family. + ["kimi-k3", "medium", "kimi-k3-high"], + ["kimi-k3", "xhigh", "kimi-k3-max"], + ["claude-opus-5", "none", "claude-opus-5-low"], + // `none` switches Thinking off where that is how the family spells it. + ["claude-sonnet-4-6", undefined, "claude-sonnet-4-6-thinking"], + ["claude-sonnet-4-6", "none", "claude-sonnet-4-6"], + ["gemini-3-5-flash", "minimal", "gemini-3-5-flash-minimal"], + // No default member: every toggle at its lowest order, effort nearest Medium. + ["gpt-5-4", undefined, "gpt-5-4-medium"], + ["glm-5-3", "medium", "glm-5-3-high"], + ["swe-1-6", undefined, "swe-1-6"], + ["swe-1-6-fast", undefined, "swe-1-6-fast"], + ["MODEL_PRIVATE_11", undefined, "MODEL_PRIVATE_11"], + ] as Array<[string, string | undefined, string]>)("%s @ %s -> %s", async (model, effort, expected) => { + expect(await resolve(catalog, model, effort)).toBe(expected); + }); + + test("a disabled member is passed over while the family has an enabled one", async () => { + const rows = LIVE_ROWS.map((r) => (r.uid === "swe-2-high" ? { ...r, disabled: true } : r)); + // High (the default) is disabled: the next enabled rung up wins. + expect(await resolve(catalogOf(rows), "swe-2")).toBe("swe-2-max"); + // Nothing enabled at or above High: the highest enabled rung below it. + const capped = rows.map((r) => (r.uid === "swe-2-max" ? { ...r, disabled: true } : r)); + expect(await resolve(catalogOf(capped), "swe-2")).toBe("swe-2-medium"); + }); + + test("an unknown caller effort keeps a named row", async () => { + expect(await resolve(catalog, "swe-2-max", "future-effort")).toBe("swe-2-max"); + }); +}); + +describe("suffix fallback without family metadata never crosses families", () => { + // Same uids, no #23/#30/#31: the path an older catalog takes. + const bare = (uids: string[]) => catalogOf(uids.map((uid) => ({ uid }))); + + test.each([ + ["claude-opus-5", ["claude-opus-5-5-low", "claude-opus-5-5-medium-fast"], "claude-opus-5-medium"], + ["gpt-5-4", ["gpt-5-4-mini-low"], "gpt-5-4-medium"], + ["glm-5-3", ["glm-5-3-flash-low"], "glm-5-3-medium"], + ] as Array<[string, string[], string]>)("%s ignores %j", async (model, uids, expected) => { + expect(await resolve(bare(uids), model)).toBe(expected); + }); + + test("a variant of the same base is still found", async () => { + expect(await resolve(bare(["claude-opus-5-5-low", "claude-opus-5-high"]), "claude-opus-5")).toBe("claude-opus-5-high"); + }); +}); + +describe("family-aware picker", () => { + let realFetch: typeof globalThis.fetch; + beforeEach(() => { + realFetch = globalThis.fetch; + globalThis.fetch = (() => { + throw new Error("devin-family-resolution.test.ts reached the network"); + }) as typeof globalThis.fetch; + setCachedCatalogForTests(catalogOf(LIVE_ROWS)); + }); + afterEach(() => { + globalThis.fetch = realFetch; + setCachedCatalogForTests(null); + }); + + test("collapses rows to families with their real ladders and default effort", async () => { + const result = await fetchDevinUsableModels({ apiKey: KEY, baseUrl: HOST }); + if (!result.ok) throw new Error(`expected ok, got ${result.error}`); + expect(result.models).toContain("swe-1-7"); + expect(result.models).toContain("claude-sonnet-4-6"); + expect(result.models).not.toContain("claude-sonnet-4-6-thinking"); + expect(result.models).not.toContain("MODEL_PRIVATE_11"); + // swe-1-6-fast is its own family, not a variant folded into swe-1-6. + expect(result.models).toContain("swe-1-6"); + expect(result.models).toContain("swe-1-6-fast"); + expect(result.efforts["swe-1-7"]).toEqual(["medium", "max"]); + expect(result.defaultEfforts["swe-1-7"]).toBe("medium"); + expect(result.efforts["swe-2"]).toEqual(["medium", "high", "max"]); + expect(result.defaultEfforts["swe-2"]).toBe("high"); + expect(result.efforts["kimi-k3"]).toEqual(["low", "high", "max"]); + expect(result.defaultEfforts["kimi-k3"]).toBe("high"); + expect(result.efforts["glm-5-2"]).toEqual(["none", "high", "max"]); + expect(result.defaultEfforts["glm-5-2"]).toBe("high"); + expect(result.efforts["gpt-6-sol"]).toEqual(["none", "low", "medium", "high", "xhigh", "max"]); + expect(result.defaultEfforts["gpt-6-sol"]).toBe("medium"); + // A family without a default member advertises no default effort. + expect(result.defaultEfforts["gpt-5-4"]).toBeUndefined(); + expect(result.inputModalities["swe-1-6"]).toEqual(["text"]); + expect(result.inputModalities["swe-1-6-fast"]).toEqual(["text", "image"]); + }); +}); diff --git a/tests/providers/devin-hardening.test.ts b/tests/providers/devin-hardening.test.ts index 2f8868e4890..e7d5e575a00 100644 --- a/tests/providers/devin-hardening.test.ts +++ b/tests/providers/devin-hardening.test.ts @@ -208,7 +208,7 @@ describe("anySignal", () => { describe("devin cloud request shape", () => { // The bug this guards: #2 and #3 were swapped, so a caller asking for 32 - // output tokens wrote 32 into the context-window field and Cognition answered + // output tokens wrote 32 into #3 (max_newlines) and Cognition answered // every single turn with an opaque "an internal error occurred" - on free and // paid accounts alike. Verified on 2026-09-12 by building the same turn with a // working client and diffing the encoded messages field by field. @@ -229,12 +229,12 @@ describe("devin cloud request shape", () => { ...(completionOpts ? { completionOpts } : {}), }); - test("the output cap lands in #2 and the context window in #3", () => { - const outer = fields(build({ maxOutputTokens: 64, maxInputTokens: 200_000 })); + test("the output cap lands in #2 and max_newlines in #3", () => { + const outer = fields(build({ maxOutputTokens: 64 })); const completion = outer[8]?.value as Buffer; const inner = fields(completion); expect(inner[2]).toEqual({ wire: 0, value: 64n }); - expect(inner[3]).toEqual({ wire: 0, value: 200_000n }); + expect(inner[3]).toEqual({ wire: 0, value: 128_000n }); // #6 and #11 are not part of the message the service accepts. expect(inner[6]).toBeUndefined(); expect(inner[11]).toBeUndefined(); diff --git a/tests/providers/devin-live-models.test.ts b/tests/providers/devin-live-models.test.ts index 6e38da6b6f3..5341e6023dc 100644 --- a/tests/providers/devin-live-models.test.ts +++ b/tests/providers/devin-live-models.test.ts @@ -192,6 +192,26 @@ describe("devin advertised catalog input modalities", () => { expect(models[0]?.inputModalities).toEqual(["text"]); }); + test("a family's catalog default member becomes the advertised default effort", async () => { + // Real swe-1.7 shape: bare swe-1-7 is Max, swe-1-7-medium is the default. + const member = (uid: string, order: number, name: string, isDefault: boolean) => Buffer.concat([ + catalogEntry(uid), + encodeMessage(23, encodeString(23, "swe-1.7")), + encodeMessage(30, Buffer.concat([ + encodeMessage(2, Buffer.concat([ + encodeString(1, "Reasoning Effort"), + encodeMessage(2, Buffer.concat([encodeVarintField(1, order), encodeString(2, name)])), + ])), + ])), + ...(isDefault ? [encodeVarintField(31, 1)] : []), + ]); + seedCatalog(member("swe-1-7", 5, "Max", false), member("swe-1-7-medium", 2, "Medium", true)); + const models = await fetchProviderModels("devin-test", devinProvider(), 60_000); + expect(models.map((model) => model.id)).toEqual(["swe-1-7"]); + expect(models[0]?.reasoningEfforts).toEqual(["medium", "max"]); + expect(models[0]?.defaultReasoningEffort).toBe("medium"); + }); + test("a credential change cannot reuse the previous account's live roster", async () => { // The live catalog is entitlement-specific: an observation made under one // credential must not be served to the next. Before the roster cache was diff --git a/tests/providers/devin-output-budget.test.ts b/tests/providers/devin-output-budget.test.ts index ebd0b277259..45a10d0840a 100644 --- a/tests/providers/devin-output-budget.test.ts +++ b/tests/providers/devin-output-budget.test.ts @@ -13,12 +13,11 @@ import { removeTreeWithRetry } from "../helpers/remove-tree"; /** * The output budget a Devin turn actually sends. * - * CompletionConfiguration #2 is the output cap and #3 is the context window. - * Every assertion below reads both, because the defect these guard against is - * not "the number is wrong" but "the two meanings were collapsed into one": - * a caller that names no cap has to reach the configured output budget without - * the context window leaking into the field that decides how long the answer - * may run. + * CompletionConfiguration #2 is the output cap; #3 is `max_newlines`, which + * the adapter holds at a fixed large value. A caller that names no cap has to + * reach the configured budget, then the catalog's own ceiling for the selected + * row (ModelInfo #13), without the context window leaking into the field that + * decides how long the answer may run. */ describe("Devin output budget on the wire", () => { const apiKey = "ocx-devin-output-fixture"; @@ -37,11 +36,12 @@ describe("Devin output budget on the wire", () => { function fields(buf: Buffer) { return new Map([...iterFields(buf)].map(field => [field.num, field])); } - function seed(rows: Array<{ uid: string; window?: number }>): void { + function seed(rows: Array<{ uid: string; window?: number; maxOut?: number }>): void { const buffer = Buffer.concat(rows.map(row => encodeMessage(1, Buffer.concat([ encodeString(1, row.uid), encodeString(22, row.uid), ...(row.window === undefined ? [] : [encodeVarintField(18, row.window)]), + ...(row.maxOut === undefined ? [] : [encodeMessage(23, encodeVarintField(13, row.maxOut))]), encodeVarintField(4, 0), ])))); setCachedCatalogForTests(parseCatalogBuffer(buffer, apiKey, host)); @@ -62,10 +62,10 @@ describe("Devin output budget on the wire", () => { return events; } /** The completion configuration the one captured turn actually encoded. */ - function sentCompletion(): { output: bigint; context: bigint } { + function sentCompletion(): { output: bigint; newlines: bigint } { expect(requests).toHaveLength(1); const completion = fields(fields(requests[0]!).get(8)!.value as Buffer); - return { output: completion.get(2)!.value as bigint, context: completion.get(3)!.value as bigint }; + return { output: completion.get(2)!.value as bigint, newlines: completion.get(3)!.value as bigint }; } beforeEach(() => { @@ -125,12 +125,29 @@ describe("Devin output budget on the wire", () => { await run({ contextWindow: 200_000, modelContextWindows: { "swe-2-high": 180_000 } }); const sent = sentCompletion(); expect(sent.output).toBe(8192n); - expect(sent.context).toBe(180_000n); + expect(sent.newlines).toBe(128_000n); }); - test("a configured output budget does not disturb the input ceiling", async () => { + test("the selected row's catalog ceiling replaces the 8192 fallback", async () => { + seed([{ uid: "swe-2-high", window: 262_000, maxOut: 128_000 }, { uid: "swe-2-max", maxOut: 96_000 }]); + await run(); + expect(sentCompletion().output).toBe(128_000n); + }); + + test("configured budgets and an explicit cap still outrank the catalog ceiling", async () => { + seed([{ uid: "swe-2-high", maxOut: 128_000 }]); await run({ defaultMaxOutputTokens: 64_000 }); - expect(sentCompletion().context).toBe(262_000n); + expect(sentCompletion().output).toBe(64_000n); + requests = []; + await run({}, { maxOutputTokens: 500 }); + expect(sentCompletion().output).toBe(500n); + }); + + test("max_newlines stays fixed whatever the context window", async () => { + // #3 once carried each row's context window as if it were an input ceiling. + seed([{ uid: "swe-2-high", window: 1_000_000 }]); + await run({ contextWindow: 80_000 }); + expect(sentCompletion().newlines).toBe(128_000n); }); }); diff --git a/tests/providers/devin-prompt-cache.test.ts b/tests/providers/devin-prompt-cache.test.ts index 362732709ce..ea22014f2cb 100644 --- a/tests/providers/devin-prompt-cache.test.ts +++ b/tests/providers/devin-prompt-cache.test.ts @@ -1,7 +1,7 @@ import { mkdtempSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; -import { createDevinAdapter, resolveDevinMaxInputTokensForTests } from "../../src/adapters/devin"; +import { createDevinAdapter } from "../../src/adapters/devin"; import { parseCatalogBuffer, setCachedCatalogForTests } from "../../src/adapters/devin/cloud-direct/catalog"; import { encodeMessage, encodeString, encodeVarintField, iterFields } from "../../src/adapters/devin/cloud-direct/wire"; import { createTranslatorBudget } from "../../src/lib/translator-budget"; @@ -95,7 +95,7 @@ describe("session invalidation is scoped to one account", () => { }); -describe("catalog-backed input ceilings on the cached chat path", () => { +describe("one catalog read serves the cached chat path", () => { const apiKey = "ocx-devin-context-fixture"; const host = "https://server.codeium.com"; const previousHome = process.env.OPENCODEX_HOME; @@ -139,11 +139,12 @@ describe("catalog-backed input ceilings on the cached chat path", () => { event => { events.push(event); }); return events; } - function expectWire(window: number, uid = "swe-2-high"): void { + function expectWire(uid = "swe-2-high"): void { expect(requests).toHaveLength(1); const outer = fields(requests[0]!); const completion = fields(outer.get(8)!.value as Buffer); - expect(completion.get(3)!.value).toBe(BigInt(window)); + // #3 is max_newlines, fixed; no context window reaches it. + expect(completion.get(3)!.value).toBe(128_000n); expect(completion.get(2)!.value).toBe(64n); expect((outer.get(21)!.value as Buffer).toString()).toBe(uid); // A context fix must not remove prompt caching or replace the chosen model. @@ -182,54 +183,29 @@ describe("catalog-backed input ceilings on the cached chat path", () => { removeTreeWithRetry(home); }); - test.each([262_000, 1_000_000])("forwards the account's %i input ceiling, not 128k", async window => { + test.each([262_000, 1_000_000])("a seeded catalog (window %i) serves the turn without a refetch", async window => { seed([{ uid: "swe-2-high", window }]); const events = await run(); expect(events.some(event => event.type === "error")).toBe(false); expect(events).toContainEqual({ type: "text_delta", text: "ok" }); expect(events.at(-1)?.type).toBe("done"); - expectWire(window); + expectWire(); expect(urls).toHaveLength(1); // Seeded metadata stays cached through preflight. }); - test.each([ - [{ contextWindow: 80_000 }, 80_000], - [{ modelContextWindows: { "swe-2": 90_000 } }, 90_000], - [{ modelContextWindows: { "swe-2-high": 100_000, "swe-2": 180_000 } }, 100_000], - [{ modelContextWindows: { "swe-2": 1_000_000 } }, 262_000], - [{ modelMaxInputTokens: { "swe-2": 70_000 }, contextWindow: 90_000 }, 70_000], - [{ modelContextWindows: { "SWE.2": 110_000 } }, 110_000], - ] as Array<[Partial, number]>)("preserves smaller configured hints %j", async (provider, expected) => { - seed([{ uid: "swe-2-high", window: 262_000 }]); - await run("swe-2-high", provider); - expectWire(expected); - }); - - test.each(["gpt-5-6-sol-high", "gpt-5-6-sol-high-1m"])("uses exact variant evidence for %s", async uid => { + test.each(["gpt-5-6-sol-high", "gpt-5-6-sol-high-1m"])("sends the exact variant %s", async uid => { seed([ { uid: "gpt-5-6-sol-high", window: 200_000 }, { uid: "gpt-5-6-sol-high-1m", window: 1_000_000 }, ]); await run(uid); - expectWire(uid.endsWith("-1m") ? 1_000_000 : 200_000, uid); + expectWire(uid); }); - test("looks up the final effort UID rather than the originally requested variant", async () => { + test("resolves the final effort UID rather than the originally requested variant", async () => { seed([{ uid: "swe-2-medium", window: 240_000 }, { uid: "swe-2-high", window: 262_000 }]); await run("devin/swe-2-high", {}, { reasoning: "medium" }); - expectWire(240_000, "swe-2-medium"); - }); - - test.each([undefined, 0])("keeps 128k when the exact row has no positive window (%p)", async window => { - seed([{ uid: "swe-2-high", window }, { uid: "swe-2-max", window: 1_000_000 }]); - await run(); - expectWire(128_000); - }); - - test("falls back to the operator's input hint when discovery is unavailable", async () => { - await run("swe-2-high", { modelMaxInputTokens: { "swe-2": 60_000 } }); - expectWire(60_000); - expect(urls.filter(url => url.endsWith("/GetChatMessage"))).toHaveLength(1); + expectWire("swe-2-medium"); }); test("a failed catalog lookup is not retried within the turn", async () => { @@ -243,11 +219,6 @@ describe("catalog-backed input ceilings on the cached chat path", () => { expect(urls.filter(url => url.endsWith("/GetChatMessage"))).toHaveLength(1); }); - test("keeps the encoder default without discovery or a configured hint", async () => { - await run(); - expectWire(128_000); - }); - test.each([true, false])("retains disabled/unlisted preflight rejection (%p)", async disabled => { seed([{ uid: disabled ? "swe-2-high" : "other-high", window: 262_000, disabled }]); const events = await run(); @@ -263,17 +234,3 @@ describe("catalog-backed input ceilings on the cached chat path", () => { expect(urls).toHaveLength(0); }); }); - -describe("Devin input ceiling validation", () => { - test.each([NaN, Infinity, -Infinity, 0, -1, 1.5, Number.MAX_SAFE_INTEGER + 1])("ignores invalid numeric metadata %p", invalid => { - const provider = { adapter: "devin", contextWindow: invalid, modelMaxInputTokens: { "swe-2": invalid } }; - expect(resolveDevinMaxInputTokensForTests(provider, "swe-2-high", invalid)).toBeUndefined(); - expect(resolveDevinMaxInputTokensForTests(provider, "swe-2-high", 262_000)).toBe(262_000); - }); - - test("does not borrow another variant's input limit", () => { - expect(resolveDevinMaxInputTokensForTests({ - adapter: "devin", modelContextWindows: { "swe-2-max": 1_000_000 }, - }, "swe-2-high")).toBeUndefined(); - }); -});