diff --git a/docs-site/src/content/docs/guides/codex-integration.md b/docs-site/src/content/docs/guides/codex-integration.md index dd19359abf9..e8b940b202d 100644 --- a/docs-site/src/content/docs/guides/codex-integration.md +++ b/docs-site/src/content/docs/guides/codex-integration.md @@ -522,9 +522,15 @@ change sign-in, the conversation's ordinary model, Codex's provider ID, or the d Recovery does not replay after cancellation, semantic output, tool side effects, an exhausted send budget, or an authentication, admission or policy refusal. Generic `400` errors do not -enable fallback. The separately opted-in Devin `invalid_argument` case applies only to an -identified compaction failure from that adapter. The emergency attempt shares the original -request's send budget and never starts a second recovery attempt. +enable fallback. Devin answers an oversized history with an opaque pre-output +`invalid_argument`; when the request's estimated size is at or near the model's input window, +the adapter reports it as `context_length_exceeded` instead, so Codex compacts on an ordinary +turn and a failed compaction qualifies as a context overflow without any Devin-specific option. +The estimate uses tool descriptions after Cognition sanitization and truncation, matching the +request sent upstream. A smaller request that gets the same code stays a plain `400`. The +separately opted-in `allowDevinInvalidArgument` case covers only those remaining `invalid_argument` failures, and +only on an identified compaction request. The emergency attempt shares the original request's +send budget and never starts a second recovery attempt. Native encrypted compaction is outside this recovery path: its original error is retained. There is no automatic local truncation mode. A response being accepted is not proof that a diff --git a/scripts/test-layout/layout.json b/scripts/test-layout/layout.json index 2a6ea52d970..e49fac447b3 100644 --- a/scripts/test-layout/layout.json +++ b/scripts/test-layout/layout.json @@ -746,6 +746,7 @@ "destination-policy-resolved.test.ts": "routing", "devin-adapter-reset-wait.test.ts": "providers", "devin-adapter.test.ts": "providers", + "devin-chat-wire-fixes.test.ts": "providers", "devin-cli-authmode-migration.test.ts": "providers", "devin-effort-ladder.test.ts": "providers", "devin-family-resolution.test.ts": "providers", diff --git a/src/adapters/devin.ts b/src/adapters/devin.ts index 73e0b4c1a51..362cafd4c33 100644 --- a/src/adapters/devin.ts +++ b/src/adapters/devin.ts @@ -17,6 +17,7 @@ import { buildNonOpenAIToolCatalogNudgeForTools } from "./tool-catalog-nudge"; import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiServer } from "../oauth/devin"; import { isProviderIssuedThinkingSignature } from "../responses/reasoning-envelope"; import { SendBudgetExhaustedError } from "../lib/upstream-retry"; +import { devinContextOverflowEvent, isDevinHistoryOverflow } from "./devin/context-overflow"; /** * Combine two usage frames from one turn by keeping the larger count per field. @@ -346,6 +347,25 @@ function resolveDevinMaxOutputTokens( /** Pure test seam; runtime uses the same resolver immediately before dispatch. */ export const resolveDevinMaxOutputTokensForTests = resolveDevinMaxOutputTokens; +/** The classifier reads the selected UID's catalog input window, capped by configured limits. */ +function resolveDevinContextWindow( + provider: OcxProviderConfig, + modelUid: string, + catalogRow?: Pick, +): number | undefined { + const familyBase = catalogRow?.familyUid ? devinFamilyBaseId(catalogRow.familyUid) : undefined; + const limits = [ + positiveTokenCount(catalogRow?.contextWindow), + devinModelTokenHint(provider.modelContextWindows, modelUid, familyBase), + positiveTokenCount(provider.contextWindow), + devinModelTokenHint(provider.modelMaxInputTokens, modelUid, familyBase), + ].filter((value): value is number => value !== undefined); + return limits.length > 0 ? Math.min(...limits) : undefined; +} + +/** Pure test seam for configured caps and selected-row lookup. */ +export const resolveDevinContextWindowForTests = resolveDevinContextWindow; + export class DevinMissingCredentialError extends Error { constructor() { super("Devin live transport requires a Devin API key. Run ocx login devin to sign in with your Cognition/Devin account."); @@ -519,6 +539,10 @@ function mapOneMessage(message: OcxMessage): ChatHistoryItem | undefined { }; } if (message.role === "toolResult") { + // #9 alone is not enough: live, with a neutral "hello world" result flagged + // as an error, only gemini-3-8-flash reported a failure; swe-1-6, + // gpt-6-sol-low and gpt-5-6-luna-low read it as success. So the flag rides + // with the in-band marker rather than replacing it. const wireContent = mapOcxContentToWire(message.content); const toolContent = message.isError ? (typeof wireContent === "string" @@ -529,6 +553,7 @@ function mapOneMessage(message: OcxMessage): ChatHistoryItem | undefined { role: "tool", content: toolContent, tool_call_id: message.toolCallId, + ...(message.isError ? { is_error: true } : {}), }; } return undefined; @@ -695,6 +720,11 @@ export function createDevinAdapter( let openToolId: string | undefined; let usage: OcxUsage | undefined; let stopReason: string | undefined; + // Kept outside the try so the catch can tell an oversized history from a bad request. + let producedOutput = false; + let contextWindow: number | undefined; + let messages: ChatHistoryItem[] = []; + let tools: ToolDef[] | undefined; const closeOpenTool = () => { if (!openToolId) return; @@ -704,6 +734,9 @@ export function createDevinAdapter( try { // Read the selected UID's catalog row, not the picker's collapsed base. + contextWindow = resolveDevinContextWindow(provider, modelUid, catalog?.byUid.get(modelUid)); + messages = mapOcxMessagesToDevin(parsed); + tools = mapOcxToolsToDevin(parsed.context.tools); const maxOutputTokens = resolveDevinMaxOutputTokens( provider, modelUid, parsed.options.maxOutputTokens, catalog?.byUid.get(modelUid), ); @@ -718,8 +751,8 @@ export function createDevinAdapter( apiServerUrl: host, modelUid, catalog, - messages: mapOcxMessagesToDevin(parsed), - tools: mapOcxToolsToDevin(parsed.context.tools), + messages, + tools, cascadeId, completionOpts: { ...(maxOutputTokens !== undefined ? { maxOutputTokens } : {}), @@ -746,6 +779,7 @@ export function createDevinAdapter( emit({ type: "error", message: DEVIN_CLIENT_CLOSED_MESSAGE, status: 499, retryable: false, ...(usage ? { usage } : {}) }); return; } + if (event.kind === "text" || event.kind === "reasoning" || event.kind === "tool_call_start") producedOutput = true; if (event.kind === "text") { closeOpenTool(); if (event.text) emit({ type: "text_delta", text: event.text }); @@ -818,6 +852,12 @@ export function createDevinAdapter( // The Responses boundary already maps this local refusal to its structured 429 code. // Converting it to an adapter event would make it an ordinary untyped upstream error. if (error instanceof SendBudgetExhaustedError) throw error; + if (error instanceof CloudChatError && isDevinHistoryOverflow({ + code: error.code, producedOutput, contextWindow, messages, tools, + })) { + emit({ ...devinContextOverflowEvent(), ...(usage ? { usage } : {}) }); + return; + } const message = error instanceof CloudChatError ? ("Devin cloud error" + (error.code ? " " + error.code : "") + ": " + error.message) : error instanceof Error ? error.message : String(error); diff --git a/src/adapters/devin/cloud-direct/chat.ts b/src/adapters/devin/cloud-direct/chat.ts index 1103ac9db99..bff84b113ec 100644 --- a/src/adapters/devin/cloud-direct/chat.ts +++ b/src/adapters/devin/cloud-direct/chat.ts @@ -39,6 +39,7 @@ import { getCachedCatalog, ModelNotAvailableError, type CacheEntry } from './cat import { anySignal, cancelBodyOnAbort } from '../../../lib/abort.js'; import { parseRetryAfterFromMessage } from '../../../lib/retry-delay.js'; import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js'; +import { normalizeDevinToolParameters } from './tool-schema.js'; /** * Connect-RPC streaming inactivity timeout. If the cloud sends zero bytes @@ -175,6 +176,7 @@ export function allocateCascadeId(): string { * #3 prompt: string (text content) * #4 num_tokens: int (rough estimate) * #5 safe_for_code_telemetry: bool (1 = ok to log) + * #9 tool_result_is_error: bool (tool prompts only) * #10 images: repeated ImageData (multimodal) * #11 thinking: string (assistant reasoning, replayed) * #12 signature: string (opaque attestation for #11) @@ -218,6 +220,7 @@ function encodeChatMessagePrompt( thinking?: string; signature?: string; signatureType?: string; + isError?: boolean; }, ): Buffer { const textParts = content.filter((p): p is { type: 'text'; text: string } => p.type === 'text'); @@ -234,6 +237,9 @@ function encodeChatMessagePrompt( if (opts?.toolCallId) { parts.push(encodeString(7, opts.toolCallId)); } + // Accepted live on a tool prompt. Only some models act on it, so the adapter + // also keeps an in-band marker in the text. + if (opts?.isError) parts.push(encodeVarintField(9, 1)); // Assistant message with tool_calls: encode each as a ChatToolCall. if (opts?.toolCalls && opts.toolCalls.length > 0) { for (const tc of opts.toolCalls) { @@ -257,30 +263,26 @@ function encodeChatMessagePrompt( const SOURCE_BY_ROLE: Record = { user: 1, assistant: 2, - // NOTE: do not send source=3 (SYSTEM) directly — the Codeium chat backend - // returns "third-party model provider is experiencing issues" when any - // ChatMessagePrompt has source=SYSTEM. The captured LS upstream traffic - // shows the IDE inlines system context into the *user* prompt (source=1) - // wrapped in .... We collapse - // role:'system' messages into the next user turn before building the - // proto — see `collapseSystemIntoUser` below. + // Never sent as a prompt source: the leading system text goes in request #2, + // and a later system message is collapsed into the next user turn below. system: 1, tool: 4, }; /** - * Collapse OpenAI-style messages so all `role:'system'` entries are inlined - * into the immediately-following user message, matching the wire format the - * IDE uses. Cognition's chat backend rejects raw role=system entries. + * Collapse `role:'system'` entries that follow the conversation start into the + * immediately-following user message. The leading run of system messages never + * reaches here; it is the request's #2 system prompt. With S0 already sent as #2: * - * [{system: "S1"}, {system: "S2"}, {user: "U1"}, {assistant: "A1"}, {user: "U2"}] + * [{user: "U1"}, {assistant: "A1"}, {system: "S1"}, {system: "S2"}, {user: "U2"}] * * becomes * - * [{user: "\nS1\nS2\n\nU1"}, {assistant: "A1"}, {user: "U2"}] + * [{user: "U1"}, {assistant: "A1"}, {user: "\nS1\n\nS2\n\nU2"}] * * If there's no following user message, the trailing system messages get - * appended as a synthesized user turn. + * appended as a synthesized user turn. A request made only of system messages + * keeps no #2 and comes through here whole, so its prompt list is never empty. */ function collapseSystemIntoUser(messages: ChatHistoryItem[]): ChatHistoryItem[] { const out: ChatHistoryItem[] = []; @@ -428,6 +430,8 @@ export interface ChatHistoryItem { thinking?: string; signature?: string; signature_type?: string; + /** For `role: 'tool'` only — the tool failed. Encoded as ChatMessagePrompt #9. */ + is_error?: boolean; } /** @@ -535,8 +539,8 @@ interface BuildArgs { messages: ChatHistoryItem[]; cascadeId: string; /** - * GetChatMessageRequest #22. Optional because it is omitted on a first turn; - * the working client only reuses one across a later tool loop. + * GetChatMessageRequest #17 prompt_id. Optional because it is omitted on a + * first turn; the working client only reuses one across a later tool loop. */ promptId?: string; sessionId: string; @@ -639,16 +643,20 @@ export function sanitizeToolDescriptionForCognitionForTests(description: string) return sanitizeToolDescriptionForCognition(description); } -function encodeToolDef(tool: ToolDef): Buffer { - const rawDesc = sanitizeToolDescriptionForCognition(tool.description ?? ''); - const desc = - rawDesc.length > MAX_TOOL_DESC_LEN - ? rawDesc.slice(0, MAX_TOOL_DESC_LEN - 24) + '\n…(truncated for cloud)' - : rawDesc; +/** Description as transmitted on the Cognition wire, also used by overflow estimation. */ +export function prepareToolDescriptionForCognition(description: string): string { + const rawDesc = sanitizeToolDescriptionForCognition(description); + return rawDesc.length > MAX_TOOL_DESC_LEN + ? rawDesc.slice(0, MAX_TOOL_DESC_LEN - 24) + '\n…(truncated for cloud)' + : rawDesc; +} + +function encodeToolDef(tool: ToolDef, modelUid: string): Buffer { + const desc = prepareToolDescriptionForCognition(tool.description ?? ''); return Buffer.concat([ encodeString(1, tool.name), encodeString(2, desc), - encodeString(3, JSON.stringify(tool.parameters ?? {})), + encodeString(3, JSON.stringify(normalizeDevinToolParameters(modelUid, tool.parameters ?? {}))), ]); } @@ -667,9 +675,22 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer { cloudChatShape: true, }); - // System messages must be inlined into the user turn (Cognition cloud - // rejects source=3). See `collapseSystemIntoUser` for the format. - const collapsed = collapseSystemIntoUser(args.messages); + // The leading system messages become request #2. Measured live on swe-1-6 + // with a ~6.7k-token system prompt: turn 2 read 6688 of 6715 prompt tokens + // from cache in #2, against 7072 of 7097 when the same text was collapsed + // into the first user prompt, so the cache ratio is unchanged and the prompt + // is smaller. The model obeyed an instruction given only in #2. + // A request with only system text keeps it as a user prompt: #2 alone would + // leave the request with no prompt at all. + const firstNonSystem = args.messages.findIndex((m) => m.role !== 'system'); + const leadingSystem = firstNonSystem === -1 ? [] : args.messages.slice(0, firstNonSystem); + const systemPrompt = leadingSystem + .map((m) => normalizeContent(m.content) + .filter((p): p is { type: 'text'; text: string } => p.type === 'text') + .map((p) => p.text).join('\n')) + .filter(Boolean) + .join('\n\n'); + const collapsed = collapseSystemIntoUser(args.messages.slice(leadingSystem.length)); const promptParts = collapsed.map((m) => encodeMessage( 3, @@ -686,6 +707,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer { thinking: m.role === 'assistant' ? m.thinking : undefined, signature: m.role === 'assistant' ? m.signature : undefined, signatureType: m.role === 'assistant' ? m.signature_type : undefined, + isError: m.role === 'tool' ? m.is_error : undefined, }, ), ), @@ -694,7 +716,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer { const completion = encodeCompletionConfiguration(args.completionOpts ?? {}); const toolParts: Buffer[] = (args.tools ?? []).map((t) => - encodeMessage(10, encodeToolDef(t)), + encodeMessage(10, encodeToolDef(t, args.modelUid)), ); // Field layout from mitm capture of the LS: @@ -704,15 +726,15 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer { // #8 completion_configuration // #10 tools (repeated ChatToolDefinition) // #13 prompt_cache_options + // #15 CortexTrajectoryReference // #16 cascade_id (string) + // #17 prompt_id (string) // #21 chat_model_uid (string) - // #22 prompt_id (string) + // #22 execution_id (string) return Buffer.concat([ encodeMessage(1, metadata), - // #2 system_prompt is always written, empty when the caller had none. The - // system turn is separately collapsed into the first user message because - // source=SYSTEM is refused; this field is the one the wire expects here. - encodeString(2, ''), + // #2 system_prompt is always written, empty when the caller had none. + encodeString(2, systemPrompt), ...promptParts, encodeVarintField(7, args.requestType ?? 5), encodeMessage(8, completion), @@ -724,7 +746,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer { // it and records real savings; sending it unconditionally matches both the // native client and CLIProxyAPIPlus, which places it outside its tools gate. encodeMessage(13, encodeVarintField(1, PROMPT_CACHE_EPHEMERAL)), - // #15 session model config: { id, turn, 4 }. Present on every verified + // #15 CortexTrajectoryReference: { id, 1, 4 }. Present on every verified // request. encodeMessage(15, Buffer.concat([ encodeString(1, crypto.randomUUID()), @@ -734,9 +756,9 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer { encodeString(16, args.cascadeId), encodeVarintField(20, 1), encodeString(21, args.modelUid), - // #22 is deliberately omitted. It is a user-exchange id that only appears - // from the second turn onward and is reused across that turn's tool loop; a - // fresh per-request uuid matches neither shape. + // #17 prompt_id is deliberately omitted. It is a user-exchange id that only + // appears from the second turn onward and is reused across that turn's tool + // loop; a fresh per-request uuid matches neither shape. ]); } diff --git a/src/adapters/devin/cloud-direct/tool-schema.ts b/src/adapters/devin/cloud-direct/tool-schema.ts new file mode 100644 index 00000000000..3fda617bd77 --- /dev/null +++ b/src/adapters/devin/cloud-direct/tool-schema.ts @@ -0,0 +1,121 @@ +/** + * Tool-schema compatibility for Gemini models behind GetChatMessage. + * + * Measured live on gemini-3-8-flash-medium: any schema node whose `type` is a + * JSON-Schema type array (`["string", "null"]`) is refused with an opaque + * `invalid_argument` on every turn, while the same union spelled as + * `anyOf: [{type: "string"}, {type: "null"}]` is accepted, as are `$schema`, + * `additionalProperties: false`, `const`, and `$ref`/`$defs`. Claude models + * accept type arrays, so only the Gemini family is rewritten, and only that one + * keyword: the full Google subset sanitizer would strip keywords this backend + * accepts. + */ + +export function isDevinGeminiModelUid(modelUid: string): boolean { + return /^gemini-/i.test(modelUid) || /^MODEL_GOOGLE_GEMINI_/i.test(modelUid); +} + +/** Keys whose values are instance data, not subschemas. */ +const DATA_KEYS = new Set(['enum', 'const', 'default', 'examples', 'example']); +/** Keys whose values map arbitrary names (which may be "enum" or "default") to subschemas. */ +const SCHEMA_MAP_KEYS = new Set(['properties', 'patternProperties', '$defs', 'definitions', 'dependentSchemas', 'dependencies']); +/** Keys that describe the whole node and stay on it rather than moving into each branch. */ +const ANNOTATION_KEYS = new Set(['description', 'title', 'default', 'examples', 'example', '$comment', 'deprecated']); + +function rewriteMap(map: unknown): unknown { + if (!map || typeof map !== 'object' || Array.isArray(map)) return map; + // A draft-7 `dependencies` entry may be a list of property names, which is data. + return Object.fromEntries(Object.entries(map).map(([name, schema]) => [name, Array.isArray(schema) ? schema : rewrite(schema)])); +} + +type Schema = Record; +const isSchema = (value: unknown): value is Schema => !!value && typeof value === 'object' && !Array.isArray(value); + +/** + * `{type: [T, "null"], ...rest}` becomes `{anyOf: [{...rest, type: T}, {type: "null"}]}`, + * so type-specific keywords (items, properties, ...) stay attached to their type. + * An existing anyOf is folded in branch by branch when no branch contradicts an + * outer keyword, and kept beside the split under allOf when one does. + */ +function splitTypeArray(node: Schema, types: unknown[]): Schema { + const annotations: Schema = {}; + const rest: Schema = {}; + for (const [key, value] of Object.entries(node)) { + if (key === 'type' || key === 'anyOf') continue; + Object.defineProperty(ANNOTATION_KEYS.has(key) ? annotations : rest, key, { value, enumerable: true, writable: true, configurable: true }); + } + const concrete = types.filter((type) => type !== 'null'); + // An outer enum/const is still binding after the type array is split. + const allowsNull = concrete.length < types.length + && (!Array.isArray(rest.enum) || rest.enum.includes(null)) + && (!Object.hasOwn(rest, 'const') || rest.const === null); + const existing = Array.isArray(node.anyOf) ? node.anyOf : undefined; + // Boolean constraints apply to every type, including null. A bare null branch + // would bypass them, so keep the type split and the outer schema conjunctive. + if (allowsNull && ['not', 'allOf', 'oneOf', 'if', 'then', 'else'].some((key) => key in rest)) { + return { + ...annotations, + allOf: [splitTypeArray({ type: types }, types), existing ? { ...rest, anyOf: existing } : rest], + }; + } + // Folding merges each branch into the outer keywords, which is only exact when + // they never disagree: `{maxLength: 5, anyOf: [{maxLength: 50}]}` folded would + // loosen the outer limit. On a disagreement keep both constraints under allOf, + // which this backend accepts (live: gemini-3-8-flash-medium). + if (existing?.some((branch) => isSchema(branch) && ( + Object.keys(branch).some((key) => key !== 'type' && key in rest && JSON.stringify(branch[key]) !== JSON.stringify(rest[key])) + // Keep an existing null branch's own constraints; folding it to {type:"null"} + // would admit values that its enum, const, or nested schema rejects. + || (allowsNull && branch.type === 'null' && Object.keys(branch).some((key) => key !== 'type')) + || (allowsNull && branch.type === undefined && ['enum', 'const', 'not', 'allOf', 'oneOf'].some((key) => key in branch)) + ))) { + return { ...annotations, allOf: [splitTypeArray({ ...rest, type: types }, types), { anyOf: existing }] }; + } + const branches: unknown[] = []; + let nullReachable = allowsNull && !existing; + if (!existing) { + for (const type of concrete) branches.push({ ...rest, type }); + } else { + for (const branch of existing) { + if (!isSchema(branch)) continue; + if (branch.type === undefined) { + for (const type of concrete) branches.push({ ...rest, ...branch, type }); + if (allowsNull) nullReachable = true; + } else if (branch.type === 'null') { + if (allowsNull) nullReachable = true; + } else if (concrete.includes(branch.type)) { + branches.push({ ...rest, ...branch }); + } + } + } + if (nullReachable) branches.push({ type: 'null' }); + // The two unions are disjoint. Keep both constraints rather than drop the type union, + // which would let the anyOf branches admit types the node never allowed. + if (branches.length === 0) return { ...annotations, allOf: [splitTypeArray({ ...rest, type: types }, types), { anyOf: existing }] }; + if (branches.length === 1 && isSchema(branches[0])) return { ...annotations, ...branches[0] }; + return { ...annotations, anyOf: branches }; +} + +function rewrite(node: unknown): unknown { + if (Array.isArray(node)) return node.map(rewrite); + if (!isSchema(node)) return node; + // fromEntries defines own properties, so a "__proto__" key stays data. + const out: Schema = Object.fromEntries( + Object.entries(node).map(([key, value]) => [ + key, + DATA_KEYS.has(key) ? value : SCHEMA_MAP_KEYS.has(key) ? rewriteMap(value) : rewrite(value), + ]), + ); + if (!Array.isArray(out.type)) return out; + const types = out.type as unknown[]; + if (types.length === 0) { + delete out.type; + return out; + } + return splitTypeArray(out, types); +} + +/** Rewrite type arrays for Gemini uids; every other model gets the schema unchanged. */ +export function normalizeDevinToolParameters(modelUid: string, parameters: unknown): unknown { + return isDevinGeminiModelUid(modelUid) ? rewrite(parameters) : parameters; +} diff --git a/src/adapters/devin/context-overflow.ts b/src/adapters/devin/context-overflow.ts new file mode 100644 index 00000000000..21cea7729ad --- /dev/null +++ b/src/adapters/devin/context-overflow.ts @@ -0,0 +1,75 @@ +/** + * Recognise an oversized Devin history behind Cognition's opaque refusal. + * + * Measured live on swe-1-6 (200k context): a 1.2 MB or 3 MB history is refused + * after ~11s with `invalid_argument` and no output, the same code a bad tool + * schema gets. Passed through as a plain 400, Codex never learns its context is + * full and never compacts, so every later turn dead-ends the same way. Only a + * request large against the model's window is reclassified; a small request + * with the same code stays an ordinary 400. + * + * Boundary, binary-searched live on swe-1-6 (200k catalog window, 8192 output): + * accepted at 199,186 prompt tokens of prose and 200,345 of JSON, refused at + * ~203k of either. Raising the output cap to 32000 did not move it (195,571 + * input tokens still accepted), so the window is not reduced by the output + * reservation and the threshold is the window itself. + */ +import type { AdapterEvent } from "../../types"; +import type { ChatHistoryItem, ToolDef } from "./cloud-direct"; +import { prepareToolDescriptionForCognition } from "./cloud-direct/chat"; + +/** + * Share of the window the estimate must reach. Characters per real token ran + * from 1.33 (Korean) through 2.31 (JSON) and 4.23 (source code) to 5.53 + * (prose), so no character ratio separates "over the window" from "60% of it". + * The word-piece count below read 1.00x to 1.54x of the real count on those + * samples: at 0.95 every sample at the window is caught and none at 60% is. + */ +const WINDOW_SHARE = 0.95; +/** One token per short letter run, 1-3 digit group, or other visible character. */ +const WORD_PIECE = /[A-Z]?[a-z]{1,8}|[A-Z]{1,8}(?![a-z])|\d{1,3}|\S/g; +/** Text size that counts as oversized when the model's window is unknown. */ +const UNKNOWN_WINDOW_CHARS = 512 * 1024; + +export const DEVIN_CONTEXT_OVERFLOW_MESSAGE = + "Devin rejected this turn and its history is at or past the model's context window. Compact the conversation or start a new session."; + +/** Text the model would tokenize; image bytes are excluded so a screenshot cannot look like a long history. */ +function requestText(messages: ChatHistoryItem[], tools: ToolDef[] | undefined): string { + const parts: string[] = []; + for (const m of messages) { + if (typeof m.content === "string") parts.push(m.content); + else for (const p of m.content) if (p.type === "text") parts.push(p.text); + for (const call of m.tool_calls ?? []) parts.push(call.arguments); + if (m.thinking) parts.push(m.thinking); + } + for (const tool of tools ?? []) parts.push(prepareToolDescriptionForCognition(tool.description ?? ""), JSON.stringify(tool.parameters ?? {})); + return parts.join("\n"); +} + +export function isDevinHistoryOverflow(input: { + code: string | undefined; + producedOutput: boolean; + contextWindow: number | undefined; + messages: ChatHistoryItem[]; + tools: ToolDef[] | undefined; +}): boolean { + if (input.code !== "invalid_argument" || input.producedOutput) return false; + const text = requestText(input.messages, input.tools); + if (!input.contextWindow) return text.length >= UNKNOWN_WINDOW_CHARS; + let pieces = 0; + for (const _ of text.matchAll(WORD_PIECE)) pieces++; + return pieces >= input.contextWindow * WINDOW_SHARE; +} + +/** Same terminal shape the Kiro adapter uses, which Codex reads as "context full, compact". */ +export function devinContextOverflowEvent(): Extract { + return { + type: "error", + message: DEVIN_CONTEXT_OVERFLOW_MESSAGE, + status: 400, + errorType: "invalid_request_error", + code: "context_length_exceeded", + retryable: false, + }; +} diff --git a/structure/providers-and-adapters.md b/structure/providers-and-adapters.md index d160c429c89..7c56739895f 100644 --- a/structure/providers-and-adapters.md +++ b/structure/providers-and-adapters.md @@ -138,7 +138,7 @@ rewrite rules and the routed-id settlement. | `src/adapters/declaration-carrier.ts`, `src/adapters/input-media-guard.ts` | Default-deny allowlists for constraints the normalized request carries but a wire may not be able to express: `tools[*].allowed_callers`, which fences a tool off from callers, and inline document bytes. Both are refused with a 400 at the single guard every registered adapter passes through, rather than left to each adapter, because an adapter that never learned about the carrier rebuilds without it and answers normally. `allowed_callers` reaches the `anthropic` wire; document bytes reach `anthropic`, `openai-chat` and `google`; the `openai-responses` wire is exempt from the whole guard because it forwards the original body. Adding an `AdapterWire` member makes the omission visible in these lists instead of at a customer's upstream. The unrestricted `["direct"]` caller default is not a restriction. | | `src/adapters/azure.ts` | Azure OpenAI bridge. | | `src/adapters/cursor.ts`, `src/adapters/cursor/` | Cursor protobuf transport: discovery, request builder, event decoding, MCP, thread continuity, native-exec policy. | -| `src/adapters/devin.ts`, `src/adapters/devin/cloud-direct/` | Devin runTurn transport over Cognition Connect-RPC. `GetChatMessage` uses the Responses provider executor and shared physical-send budget; catalog, JWT, and `src/web-search/devin-executor.ts` native search support RPCs remain outside inference-send accounting. Provider-stated pre-output 429 reset delays are surfaced immediately by default, releasing shared active-turn capacity. A positive `OPENCODEX_DEVIN_STATED_RESET_WAIT_MS` explicitly enables bounded waiting and up to two replays on standalone turns, which hold that capacity until completion or cancellation. Combo children bypass that wait and surface a pre-output 429 so the next target can run. During an opted-in standalone wait, safe heartbeats commit the response preflight and keep the stream's stall watchdog fed. Invalid values fail closed to the immediate-refusal behavior. A recorded tenant host is used only for the stored account whose credential owns the transmitted key, searched in the configured provider id and then its deprecated alias; a configured, forwarded, or unmatched key uses the configured base URL or the US default. Native search previews the current route by effective adapter without mutating combo selection state, pins one admitted active-account snapshot for the request, and calls `GetWebSearchResults`, so it starts no CLI or second model. The wire model UID comes from the catalog's family metadata (`ClientModelConfig` #23/#30/#31): a family id with no effort anchors on the family's default member and selects the nearest enabled rung, rounding up first; an effort moves only the effort axis, to the lowest rung at or above it (else the highest below) while Fast Mode, 1M Context and the other axes stay at the anchor's values unless the caller asked for `fast` or a `1m` value; not lowering a requested effort outranks keeping those axes (an unranked member counts as lower), a disabled row the caller named is kept when the request still selects it so the preflight names that refusal, and resolution never leaves the family. Rows without family metadata fall back to suffix resolution, whose variant scan matches the collapsed base rather than a string prefix. With no caller or configured output cap, the selected row's catalog `maxOutputTokens` fills CompletionConfiguration #2; #3 is `max_newlines` and is sent at a fixed value, never a context window. | +| `src/adapters/devin.ts`, `src/adapters/devin/cloud-direct/` | Devin runTurn transport over Cognition Connect-RPC. `GetChatMessage` uses the Responses provider executor and shared physical-send budget; catalog, JWT, and `src/web-search/devin-executor.ts` native search support RPCs remain outside inference-send accounting. Provider-stated pre-output 429 reset delays are surfaced immediately by default, releasing shared active-turn capacity. A positive `OPENCODEX_DEVIN_STATED_RESET_WAIT_MS` explicitly enables bounded waiting and up to two replays on standalone turns, which hold that capacity until completion or cancellation. Combo children bypass that wait and surface a pre-output 429 so the next target can run. During an opted-in standalone wait, safe heartbeats commit the response preflight and keep the stream's stall watchdog fed. Invalid values fail closed to the immediate-refusal behavior. A recorded tenant host is used only for the stored account whose credential owns the transmitted key, searched in the configured provider id and then its deprecated alias; a configured, forwarded, or unmatched key uses the configured base URL or the US default. Native search previews the current route by effective adapter without mutating combo selection state, pins one admitted active-account snapshot for the request, and calls `GetWebSearchResults`, so it starts no CLI or second model. The wire model UID comes from the catalog's family metadata (`ClientModelConfig` #23/#30/#31): a family id with no effort anchors on the family's default member and selects the nearest enabled rung, rounding up first; an effort moves only the effort axis, to the lowest rung at or above it (else the highest below) while Fast Mode, 1M Context and the other axes stay at the anchor's values unless the caller asked for `fast` or a `1m` value; not lowering a requested effort outranks keeping those axes (an unranked member counts as lower), a disabled row the caller named is kept when the request still selects it so the preflight names that refusal, and resolution never leaves the family. Rows without family metadata fall back to suffix resolution, whose variant scan matches the collapsed base rather than a string prefix. With no caller or configured output cap, the selected row's catalog `maxOutputTokens` fills CompletionConfiguration #2; #3 is `max_newlines` and is sent at a fixed value, never a context window. The leading system text is sent as `GetChatMessage` #2, a failed tool result sets ChatMessagePrompt #9 and keeps an in-band `ERROR:` marker, and Gemini uids have JSON-Schema type arrays split into per-type `anyOf` branches in tool parameters while preserving outer enum/const, boolean constraints (including `not`, `oneOf`, and `allOf`) on null, and existing null-branch restrictions. A pre-output `invalid_argument` on a history whose word-piece estimate, using the sanitized and truncated tool descriptions actually sent on the wire, reaches 95% of the selected UID's catalog input window, capped by configured provider and model limits (512 KiB of text only when no window is known) is surfaced as `context_length_exceeded` so Codex compacts; a small request with the same code stays a plain 400. Known limit: a malformed schema on a history already at or above that 95% threshold is also reported as overflow because the upstream returns the same `invalid_argument`. | | `src/adapters/kiro.ts` and `src/adapters/kiro/` | Kiro event/tool/thinking/truncation/retry handling, including an egress-aware completion fallback, fixed public HTTP 5xx text, and closed-set status/code diagnostics. The original path is a facade over leaves for wire identity, reasoning, conversation state, token estimation, payload assembly, streaming, and the adapter. | | `src/adapters/mimo-free.ts` | Mimo Free transport (client identity + JWT). Concurrent requests share one JWT bootstrap bound only to its timeout; each request stops waiting on its own abort without cancelling the others. | | `src/adapters/command-code.ts`, `src/adapters/command-code-tool-text.ts`, `src/adapters/command-code-restored-schema.ts` | Command Code OAuth NDJSON translation. For every `xiaomi/mimo-` model, text, native calls, reasoning, and terminal decisions share one byte-bounded queue with linear queue visits. Markup is deduplicated against matching native calls; text-only restoration requires one contiguous text run, a clean finish, a declared tool, and arguments validated against supported schema constraints. A parameter-free (freeform) block may omit `` but must end with ``; parameter blocks keep the canonical close. Markup appended after prose in the same delta is split off at the marker and held like a block that opens with ``; a marker split across deltas after prose is still released as text. Native, reasoning, and other intervening events interrupt a still-probing block but leave a held block held in arrival order, and the queued byte bound still flushes an unresolved envelope as text. An envelope the strict parser rejects but that opens with ``, closes with ``, and names a declared function is dropped when a native call for that same function arrives and on a clean finish; markup that parses but fits no supported schema is still released as text. Regex patterns, other unsupported constraints, and abnormal finishes fail closed. `tests/providers/command-code-tool-text-prose-split.test.ts` covers the split, the interleaved-event hold, and both drop paths. | diff --git a/tests/fixtures/test-layout-expected.json b/tests/fixtures/test-layout-expected.json index 622289a0b6f..2c2bab19d4c 100644 --- a/tests/fixtures/test-layout-expected.json +++ b/tests/fixtures/test-layout-expected.json @@ -751,6 +751,7 @@ "destination-policy-resolved.test.ts": "routing", "devin-adapter-reset-wait.test.ts": "providers", "devin-adapter.test.ts": "providers", + "devin-chat-wire-fixes.test.ts": "providers", "devin-cli-authmode-migration.test.ts": "providers", "devin-effort-ladder.test.ts": "providers", "devin-family-resolution.test.ts": "providers", diff --git a/tests/providers/devin-chat-wire-fixes.test.ts b/tests/providers/devin-chat-wire-fixes.test.ts new file mode 100644 index 00000000000..03b0bbcf14f --- /dev/null +++ b/tests/providers/devin-chat-wire-fixes.test.ts @@ -0,0 +1,352 @@ +/** + * GetChatMessage request shape and failure mapping, each pinned by a live + * Cognition measurement: the system prompt rides request #2, a failed tool + * result sets ChatMessagePrompt #9, Gemini tool schemas lose type arrays, and + * an oversized history's opaque invalid_argument becomes context_length_exceeded. + */ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { mkdtempSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { createDevinAdapter, mapOcxMessagesToDevin, resolveDevinContextWindowForTests } from "../../src/adapters/devin"; +import { parseCatalogBuffer, setCachedCatalogForTests } from "../../src/adapters/devin/cloud-direct/catalog"; +import { buildGetChatMessageRequestForTests, type ChatHistoryItem } from "../../src/adapters/devin/cloud-direct/chat"; +import { normalizeDevinToolParameters } from "../../src/adapters/devin/cloud-direct/tool-schema"; +import { isDevinHistoryOverflow } from "../../src/adapters/devin/context-overflow"; +import { encodeMessage, encodeString, encodeVarintField, iterFields } from "../../src/adapters/devin/cloud-direct/wire"; +import { createTranslatorBudget } from "../../src/lib/translator-budget"; +import type { AdapterEvent, OcxMessage, OcxParsedRequest } from "../../src/types"; +import { removeTreeWithRetry } from "../helpers/remove-tree"; + +function build(messages: ChatHistoryItem[], extra: Record = {}): Buffer { + return buildGetChatMessageRequestForTests({ + apiKey: "k", modelUid: "swe-1-6", messages, cascadeId: "c", sessionId: "s", requestId: 1n, triggerId: "t", ...extra, + } as never); +} + +function topFields(buf: Buffer) { + return [...iterFields(buf)]; +} + +function prompts(buf: Buffer) { + return topFields(buf).filter((f) => f.num === 3).map((f) => [...iterFields(f.value as Buffer)]); +} + +const text = (fields: ReturnType[number]) => + (fields.find((f) => f.num === 3)?.value as Buffer).toString("utf8"); + +describe("system prompt channel", () => { + test("leading system text is request #2 and no prompt carries a wrapper", () => { + const buf = build([ + { role: "system", content: "S1" }, + { role: "system", content: "S2" }, + { role: "user", content: "U1" }, + ]); + const system = topFields(buf).find((f) => f.num === 2)!.value as Buffer; + expect(system.toString("utf8")).toBe("S1\n\nS2"); + const ps = prompts(buf); + expect(ps).toHaveLength(1); + expect(text(ps[0]!)).toBe("U1"); + }); + + test("a system message after the conversation starts still reaches the model in the next user turn", () => { + const ps = prompts(build([ + { role: "system", content: "S1" }, + { role: "user", content: "U1" }, + { role: "assistant", content: "A1" }, + { role: "system", content: "LATE" }, + { role: "user", content: "U2" }, + ])); + expect(ps.map(text)).toEqual(["U1", "A1", "\nLATE\n\nU2"]); + }); + + test("a request with only system text keeps it as a user prompt", () => { + const buf = build([{ role: "system", content: "ONLY" }]); + expect((topFields(buf).find((f) => f.num === 2)!.value as Buffer).length).toBe(0); + expect(prompts(buf).map(text)).toEqual(["\nONLY\n"]); + }); + + test("#2 stays present and empty with no system text", () => { + const system = topFields(build([{ role: "user", content: "hi" }])).find((f) => f.num === 2)!; + expect((system.value as Buffer).length).toBe(0); + }); +}); + +describe("tool_result_is_error (#9)", () => { + const parsed = (isError: boolean): OcxParsedRequest => ({ + modelId: "swe-1-6", + stream: true, + context: { + messages: [ + { role: "user", content: "read it", timestamp: 1 }, + { role: "toolResult", toolCallId: "call_1", toolName: "read", content: [{ type: "text", text: "ENOENT" }], isError, timestamp: 2 } as unknown as OcxMessage, + ], + }, + options: {}, + } as OcxParsedRequest); + + test("a failed tool result sets #9=1 and keeps the in-band marker", () => { + const tool = prompts(build(mapOcxMessagesToDevin(parsed(true)))).at(-1)!; + expect(Number(tool.find((f) => f.num === 9)?.value)).toBe(1); + expect(text(tool)).toBe("ERROR:\nENOENT"); + }); + + test("a successful tool result carries no #9", () => { + const tool = prompts(build(mapOcxMessagesToDevin(parsed(false)))).at(-1)!; + expect(tool.some((f) => f.num === 9)).toBe(false); + }); +}); + +describe("Gemini tool schema type arrays", () => { + const schema = { + type: "object", + properties: { + q: { type: ["string", "null"], description: "query" }, + default: { type: ["integer", "null"] }, + tags: { type: "array", items: { type: ["string", "null"] } }, + mode: { type: "string", enum: ["a", "b"], default: "a" }, + }, + required: ["q"], + additionalProperties: false, + }; + + test("gemini uids get anyOf unions everywhere, including under a property named default", () => { + const out = normalizeDevinToolParameters("gemini-3-8-flash-medium", schema) as any; + expect(out.properties.q).toEqual({ description: "query", anyOf: [{ type: "string" }, { type: "null" }] }); + expect(out.type).toBe("object"); + expect(out.properties.default).toEqual({ anyOf: [{ type: "integer" }, { type: "null" }] }); + expect(out.properties.tags.items).toEqual({ anyOf: [{ type: "string" }, { type: "null" }] }); + expect(out.properties.mode).toEqual(schema.properties.mode); + expect(out.additionalProperties).toBe(false); + expect(JSON.stringify(out)).not.toContain('"type":['); + }); + + test("type-specific keywords stay on their typed branch", () => { + const out = normalizeDevinToolParameters("gemini-3-8-flash-medium", { + type: "object", + properties: { + list: { type: ["array", "null"], description: "d", items: { type: "string" }, minItems: 1 }, + obj: { type: ["object", "null"], properties: { a: { type: "string" } }, required: ["a"] }, + }, + }) as any; + expect(out.properties.list).toEqual({ + description: "d", anyOf: [{ type: "array", items: { type: "string" }, minItems: 1 }, { type: "null" }], + }); + expect(out.properties.obj).toEqual({ + anyOf: [{ type: "object", properties: { a: { type: "string" } }, required: ["a"] }, { type: "null" }], + }); + }); + + test("a branch that contradicts an outer keyword keeps both constraints under allOf", () => { + const out = normalizeDevinToolParameters("gemini-x", { + type: ["string", "null"], maxLength: 5, anyOf: [{ maxLength: 50 }, { type: "null" }], + }) as any; + expect(out).toEqual({ + allOf: [ + { anyOf: [{ maxLength: 5, type: "string" }, { type: "null" }] }, + { anyOf: [{ maxLength: 50 }, { type: "null" }] }, + ], + }); + }); + + test("a type union disjoint from the existing anyOf keeps both constraints", () => { + const out = normalizeDevinToolParameters("gemini-x", { + type: ["string", "null"], minLength: 2, anyOf: [{ type: "integer" }], + }) as any; + expect(out).toEqual({ + allOf: [ + { anyOf: [{ minLength: 2, type: "string" }, { type: "null" }] }, + { anyOf: [{ type: "integer" }] }, + ], + }); + }); + + test("an existing anyOf that agrees with the outer keywords is folded in, not nested under allOf", () => { + const out = normalizeDevinToolParameters("MODEL_GOOGLE_GEMINI_2_5_PRO", { + type: ["object", "null"], anyOf: [{ required: ["a"] }, { required: ["b"] }], + }) as any; + expect(out.allOf).toBeUndefined(); + expect(out.anyOf).toEqual([ + { required: ["a"], type: "object" }, { required: ["b"], type: "object" }, { type: "null" }, + ]); + const typed = normalizeDevinToolParameters("gemini-x", { + type: ["string", "null"], anyOf: [{ type: "string", format: "date" }, { type: "integer" }], + }) as any; + expect(typed).toEqual({ type: "string", format: "date" }); + }); + + test("outer enum and const exclude null from a Gemini type union", () => { + expect(normalizeDevinToolParameters("gemini-x", { type: ["string", "null"], enum: ["a"] })).toEqual({ + type: "string", enum: ["a"], + }); + expect(normalizeDevinToolParameters("gemini-x", { type: ["string", "null"], const: "a" })).toEqual({ + type: "string", const: "a", + }); + }); + + test("an existing anyOf null branch keeps its own restrictions", () => { + expect(normalizeDevinToolParameters("gemini-x", { + type: ["string", "null"], anyOf: [{ type: "string" }, { type: "null", const: "a" }], + })).toEqual({ + allOf: [ + { anyOf: [{ type: "string" }, { type: "null" }] }, + { anyOf: [{ type: "string" }, { type: "null", const: "a" }] }, + ], + }); + }); + + test("outer not and oneOf still constrain the null branch", () => { + expect(normalizeDevinToolParameters("gemini-x", { + type: ["string", "null"], not: { type: "null" }, + })).toEqual({ + allOf: [ + { anyOf: [{ type: "string" }, { type: "null" }] }, + { not: { type: "null" } }, + ], + }); + expect(normalizeDevinToolParameters("gemini-x", { + type: ["string", "null"], oneOf: [{ type: "string" }, { const: "x" }], + })).toEqual({ + allOf: [ + { anyOf: [{ type: "string" }, { type: "null" }] }, + { oneOf: [{ type: "string" }, { const: "x" }] }, + ], + }); + }); + + test("draft-7 dependencies: schema values are rewritten, name lists are left alone", () => { + const out = normalizeDevinToolParameters("gemini-x", { + type: "object", + dependencies: { a: ["b", "c"], enum: { properties: { x: { type: ["string", "null"] } } } }, + }) as any; + expect(out.dependencies.a).toEqual(["b", "c"]); + expect(out.dependencies.enum.properties.x).toEqual({ anyOf: [{ type: "string" }, { type: "null" }] }); + }); + + test("non-gemini uids and the encoded request for them are untouched", () => { + expect(normalizeDevinToolParameters("claude-sonnet-5-low", schema)).toBe(schema); + const tools = [{ name: "search", description: "d", parameters: schema }]; + expect(build([{ role: "user", content: "x" }], { tools }).includes(Buffer.from('"type":["string","null"]'))).toBe(true); + const gemini = build([{ role: "user", content: "x" }], { tools, modelUid: "gemini-3-8-flash-medium" }); + expect(gemini.includes(Buffer.from('"type":['))).toBe(false); + }); +}); + +describe("oversized history classification", () => { + // One word piece per "word ", so `words` is the estimate. + const history = (words: number): ChatHistoryItem[] => [{ role: "user", content: "word ".repeat(words) }]; + const base = { code: "invalid_argument", producedOutput: false, tools: undefined }; + + test("at the catalog window is an overflow; 60% of it is not", () => { + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: history(200_000) })).toBe(true); + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: history(120_000) })).toBe(false); + }); + + test("a malformed large schema near the window is classified as overflow", () => { + const tools = [{ name: "bad", description: "d", parameters: { unsupported: "x ".repeat(190_000) } }]; + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: history(1), tools })).toBe(true); + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: history(1), + tools: [{ ...tools[0], parameters: { unsupported: "x ".repeat(20_000) } }], + })).toBe(false); + }); + + test("a huge tool description truncated on the wire stays a plain 400", () => { + const tools = [{ name: "long", description: "word ".repeat(200_000), parameters: {} }]; + expect(build(history(1), { tools }).includes(Buffer.from("…(truncated for cloud)"))).toBe(true); + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: history(1), tools })).toBe(false); + expect(isDevinHistoryOverflow({ ...base, contextWindow: undefined, messages: history(1), tools })).toBe(false); + }); + + test("dense JSON at the window is caught even though its characters per token are low", () => { + // Measured live: 443k chars of this shape was 200,345 real tokens on swe-1-6. + let json = ""; + for (let i = 0; json.length < 443_000; i++) json += JSON.stringify({ id: i, vals: [i % 97, -i], ok: i % 3 === 0 }) + ",\n"; + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: [{ role: "user", content: json }] })).toBe(true); + }); + + test("the same code on a small request stays a plain refusal", () => { + expect(isDevinHistoryOverflow({ ...base, contextWindow: 200_000, messages: history(20_000) })).toBe(false); + }); + + test("other codes, or a turn that already produced output, are never reclassified", () => { + expect(isDevinHistoryOverflow({ ...base, code: "permission_denied", contextWindow: 200_000, messages: history(240_000) })).toBe(false); + expect(isDevinHistoryOverflow({ ...base, producedOutput: true, contextWindow: 200_000, messages: history(240_000) })).toBe(false); + }); + + test("with no known window the byte threshold decides, and image bytes do not count", () => { + expect(isDevinHistoryOverflow({ ...base, contextWindow: undefined, messages: history(120 * 1024) })).toBe(true); + expect(isDevinHistoryOverflow({ ...base, contextWindow: undefined, messages: history(20 * 1024) })).toBe(false); + const image: ChatHistoryItem[] = [{ role: "user", content: [{ type: "text", text: "see" }, { type: "image", mimeType: "image/png", base64Data: "A".repeat(2_000_000) }] }]; + expect(isDevinHistoryOverflow({ ...base, contextWindow: undefined, messages: image })).toBe(false); + }); +}); + +describe("selected Devin context window", () => { + test("the selected row supplies the window, with configured context and input caps", () => { + const row = { contextWindow: 200_000, familyUid: "swe-1-6" }; + expect(resolveDevinContextWindowForTests({ adapter: "devin", baseUrl: "" }, "swe-1-6", row)).toBe(200_000); + expect(resolveDevinContextWindowForTests({ adapter: "devin", baseUrl: "", contextWindow: 180_000, + modelContextWindows: { "swe-1-6": 170_000 }, modelMaxInputTokens: { "swe-1-6": 160_000 }, + }, "swe-1-6", row)).toBe(160_000); + expect(resolveDevinContextWindowForTests({ adapter: "devin", baseUrl: "" }, "swe-1-6")).toBeUndefined(); + }); +}); + +describe("adapter surfaces an oversized history as context_length_exceeded", () => { + const apiKey = "ocx-devin-overflow-fixture"; + const host = "https://server.codeium.com"; + const previousHome = process.env.OPENCODEX_HOME; + const previousFetch = globalThis.fetch; + let home = ""; + + beforeEach(() => { + home = mkdtempSync(join(tmpdir(), "ocx-devin-overflow-")); + process.env.OPENCODEX_HOME = home; + setCachedCatalogForTests(parseCatalogBuffer(encodeMessage(1, Buffer.concat([ + encodeString(1, "swe-1-6"), encodeString(22, "swe-1-6"), encodeVarintField(18, 200_000), encodeVarintField(4, 0), + ])), apiKey, host)); + const trailer = Buffer.from(JSON.stringify({ error: { code: "invalid_argument", message: "bad request" } })); + const frame = Buffer.alloc(5 + trailer.length); + frame[0] = 0x02; + frame.writeUInt32BE(trailer.length, 1); + trailer.copy(frame, 5); + globalThis.fetch = (async () => new Response(frame, { status: 200, headers: { "Content-Type": "application/connect+proto" } })) as unknown as typeof fetch; + }); + afterEach(() => { + globalThis.fetch = previousFetch; + setCachedCatalogForTests(null); + if (previousHome === undefined) delete process.env.OPENCODEX_HOME; + else process.env.OPENCODEX_HOME = previousHome; + removeTreeWithRetry(home); + }); + + async function turn(content: string, tools?: OcxParsedRequest["context"]["tools"]): Promise { + const events: AdapterEvent[] = []; + await createDevinAdapter({ adapter: "devin", apiKey, baseUrl: host }).runTurn!({ + modelId: "swe-1-6", stream: true, context: { messages: [{ role: "user", content, timestamp: 1 }], tools }, options: {}, + }, { headers: new Headers(), translatorBudget: createTranslatorBudget(), abortSignal: AbortSignal.timeout(5_000) }, (e) => { events.push(e); }); + return events; + } + + test("1.2 MB of history against a 200k window", async () => { + const events = await turn("word ".repeat(240_000)); + expect(events.at(-1)).toMatchObject({ type: "error", status: 400, code: "context_length_exceeded", errorType: "invalid_request_error", retryable: false }); + }); + + test("selected catalog window classifies at 95% and leaves a smaller refusal as 400", async () => { + const atThreshold = await turn("word ".repeat(190_000)); + expect(atThreshold.at(-1)).toMatchObject({ type: "error", status: 400, code: "context_length_exceeded" }); + const belowThreshold = await turn("word ".repeat(189_999)); + expect(belowThreshold.at(-1)).toMatchObject({ type: "error", status: 400, code: "invalid_argument" }); + }); + + test("a short turn with the same refusal stays invalid_argument", async () => { + const events = await turn("hi"); + expect(events.at(-1)).toMatchObject({ type: "error", status: 400, code: "invalid_argument" }); + }); + + test("a truncated huge tool description leaves invalid_argument as a plain 400", async () => { + const events = await turn("hi", [{ name: "long", description: "word ".repeat(200_000), parameters: {} }]); + expect(events.at(-1)).toMatchObject({ type: "error", status: 400, code: "invalid_argument" }); + }); +}); diff --git a/tests/providers/devin-image-passthrough.test.ts b/tests/providers/devin-image-passthrough.test.ts index b9a5ec57e54..18e6aba8274 100644 --- a/tests/providers/devin-image-passthrough.test.ts +++ b/tests/providers/devin-image-passthrough.test.ts @@ -81,7 +81,7 @@ describe("tool-result image passthrough", () => { expect(parts.some(p => p.type === "image" && p.base64Data === "iVBORw0KGgoAAAANSUhEUg")).toBe(true); }); - test("an error tool result still carries the ERROR prefix alongside images", () => { + test("an error tool result is flagged, keeps the ERROR marker, and still carries its images", () => { const items = mapOcxMessagesToDevin(parsedWith([{ role: "toolResult", toolCallId: "call_1", @@ -90,6 +90,7 @@ describe("tool-result image passthrough", () => { } as unknown as OcxMessage])); const tool = items.find(i => i.role === "tool")!; const parts = tool.content as Array>; + expect(tool.is_error).toBe(true); expect(parts[0]).toMatchObject({ type: "text", text: "ERROR:" }); expect(parts.some(p => p.type === "image")).toBe(true); });