Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 9 additions & 3 deletions docs-site/src/content/docs/guides/codex-integration.md
Original file line number Diff line number Diff line change
Expand Up @@ -522,9 +522,15 @@ change sign-in, the conversation's ordinary model, Codex's provider ID, or the d

Recovery does not replay after cancellation, semantic output, tool side effects, an exhausted
send budget, or an authentication, admission or policy refusal. Generic `400` errors do not
enable fallback. The separately opted-in Devin `invalid_argument` case applies only to an
identified compaction failure from that adapter. The emergency attempt shares the original
request's send budget and never starts a second recovery attempt.
enable fallback. Devin answers an oversized history with an opaque pre-output
`invalid_argument`; when the request's estimated size is at or near the model's input window,
the adapter reports it as `context_length_exceeded` instead, so Codex compacts on an ordinary
turn and a failed compaction qualifies as a context overflow without any Devin-specific option.
The estimate uses tool descriptions after Cognition sanitization and truncation, matching the
request sent upstream. A smaller request that gets the same code stays a plain `400`. The
separately opted-in `allowDevinInvalidArgument` case covers only those remaining `invalid_argument` failures, and
only on an identified compaction request. The emergency attempt shares the original request's
send budget and never starts a second recovery attempt.

Native encrypted compaction is outside this recovery path: its original error is retained.
There is no automatic local truncation mode. A response being accepted is not proof that a
Expand Down
1 change: 1 addition & 0 deletions scripts/test-layout/layout.json
Original file line number Diff line number Diff line change
Expand Up @@ -746,6 +746,7 @@
"destination-policy-resolved.test.ts": "routing",
"devin-adapter-reset-wait.test.ts": "providers",
"devin-adapter.test.ts": "providers",
"devin-chat-wire-fixes.test.ts": "providers",
"devin-cli-authmode-migration.test.ts": "providers",
"devin-effort-ladder.test.ts": "providers",
"devin-family-resolution.test.ts": "providers",
Expand Down
44 changes: 42 additions & 2 deletions src/adapters/devin.ts
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ import { buildNonOpenAIToolCatalogNudgeForTools } from "./tool-catalog-nudge";
import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiServer } from "../oauth/devin";
import { isProviderIssuedThinkingSignature } from "../responses/reasoning-envelope";
import { SendBudgetExhaustedError } from "../lib/upstream-retry";
import { devinContextOverflowEvent, isDevinHistoryOverflow } from "./devin/context-overflow";

/**
* Combine two usage frames from one turn by keeping the larger count per field.
Expand Down Expand Up @@ -346,6 +347,25 @@ function resolveDevinMaxOutputTokens(
/** Pure test seam; runtime uses the same resolver immediately before dispatch. */
export const resolveDevinMaxOutputTokensForTests = resolveDevinMaxOutputTokens;

/** The classifier reads the selected UID's catalog input window, capped by configured limits. */
function resolveDevinContextWindow(
provider: OcxProviderConfig,
modelUid: string,
catalogRow?: Pick<ModelCatalogEntry, "contextWindow" | "familyUid">,
): number | undefined {
const familyBase = catalogRow?.familyUid ? devinFamilyBaseId(catalogRow.familyUid) : undefined;
const limits = [
positiveTokenCount(catalogRow?.contextWindow),
devinModelTokenHint(provider.modelContextWindows, modelUid, familyBase),
positiveTokenCount(provider.contextWindow),
devinModelTokenHint(provider.modelMaxInputTokens, modelUid, familyBase),
].filter((value): value is number => value !== undefined);
return limits.length > 0 ? Math.min(...limits) : undefined;
}

/** Pure test seam for configured caps and selected-row lookup. */
export const resolveDevinContextWindowForTests = resolveDevinContextWindow;

export class DevinMissingCredentialError extends Error {
constructor() {
super("Devin live transport requires a Devin API key. Run ocx login devin to sign in with your Cognition/Devin account.");
Expand Down Expand Up @@ -519,6 +539,10 @@ function mapOneMessage(message: OcxMessage): ChatHistoryItem | undefined {
};
}
if (message.role === "toolResult") {
// #9 alone is not enough: live, with a neutral "hello world" result flagged
// as an error, only gemini-3-8-flash reported a failure; swe-1-6,
// gpt-6-sol-low and gpt-5-6-luna-low read it as success. So the flag rides
// with the in-band marker rather than replacing it.
const wireContent = mapOcxContentToWire(message.content);
const toolContent = message.isError
? (typeof wireContent === "string"
Expand All @@ -529,6 +553,7 @@ function mapOneMessage(message: OcxMessage): ChatHistoryItem | undefined {
role: "tool",
content: toolContent,
tool_call_id: message.toolCallId,
...(message.isError ? { is_error: true } : {}),
};
}
return undefined;
Expand Down Expand Up @@ -695,6 +720,11 @@ export function createDevinAdapter(
let openToolId: string | undefined;
let usage: OcxUsage | undefined;
let stopReason: string | undefined;
// Kept outside the try so the catch can tell an oversized history from a bad request.
let producedOutput = false;
let contextWindow: number | undefined;
let messages: ChatHistoryItem[] = [];
let tools: ToolDef[] | undefined;

const closeOpenTool = () => {
if (!openToolId) return;
Expand All @@ -704,6 +734,9 @@ export function createDevinAdapter(

try {
// Read the selected UID's catalog row, not the picker's collapsed base.
contextWindow = resolveDevinContextWindow(provider, modelUid, catalog?.byUid.get(modelUid));
messages = mapOcxMessagesToDevin(parsed);
tools = mapOcxToolsToDevin(parsed.context.tools);
const maxOutputTokens = resolveDevinMaxOutputTokens(
provider, modelUid, parsed.options.maxOutputTokens, catalog?.byUid.get(modelUid),
);
Expand All @@ -718,8 +751,8 @@ export function createDevinAdapter(
apiServerUrl: host,
modelUid,
catalog,
messages: mapOcxMessagesToDevin(parsed),
tools: mapOcxToolsToDevin(parsed.context.tools),
messages,
tools,
cascadeId,
completionOpts: {
...(maxOutputTokens !== undefined ? { maxOutputTokens } : {}),
Expand All @@ -746,6 +779,7 @@ export function createDevinAdapter(
emit({ type: "error", message: DEVIN_CLIENT_CLOSED_MESSAGE, status: 499, retryable: false, ...(usage ? { usage } : {}) });
return;
}
if (event.kind === "text" || event.kind === "reasoning" || event.kind === "tool_call_start") producedOutput = true;
if (event.kind === "text") {
closeOpenTool();
if (event.text) emit({ type: "text_delta", text: event.text });
Expand Down Expand Up @@ -818,6 +852,12 @@ export function createDevinAdapter(
// The Responses boundary already maps this local refusal to its structured 429 code.
// Converting it to an adapter event would make it an ordinary untyped upstream error.
if (error instanceof SendBudgetExhaustedError) throw error;
if (error instanceof CloudChatError && isDevinHistoryOverflow({
code: error.code, producedOutput, contextWindow, messages, tools,
})) {
emit({ ...devinContextOverflowEvent(), ...(usage ? { usage } : {}) });
return;
}
const message = error instanceof CloudChatError
? ("Devin cloud error" + (error.code ? " " + error.code : "") + ": " + error.message)
: error instanceof Error ? error.message : String(error);
Expand Down
92 changes: 57 additions & 35 deletions src/adapters/devin/cloud-direct/chat.ts
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,7 @@ import { getCachedCatalog, ModelNotAvailableError, type CacheEntry } from './cat
import { anySignal, cancelBodyOnAbort } from '../../../lib/abort.js';
import { parseRetryAfterFromMessage } from '../../../lib/retry-delay.js';
import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js';
import { normalizeDevinToolParameters } from './tool-schema.js';

/**
* Connect-RPC streaming inactivity timeout. If the cloud sends zero bytes
Expand Down Expand Up @@ -175,6 +176,7 @@ export function allocateCascadeId(): string {
* #3 prompt: string (text content)
* #4 num_tokens: int (rough estimate)
* #5 safe_for_code_telemetry: bool (1 = ok to log)
* #9 tool_result_is_error: bool (tool prompts only)
* #10 images: repeated ImageData (multimodal)
* #11 thinking: string (assistant reasoning, replayed)
* #12 signature: string (opaque attestation for #11)
Expand Down Expand Up @@ -218,6 +220,7 @@ function encodeChatMessagePrompt(
thinking?: string;
signature?: string;
signatureType?: string;
isError?: boolean;
},
): Buffer {
const textParts = content.filter((p): p is { type: 'text'; text: string } => p.type === 'text');
Expand All @@ -234,6 +237,9 @@ function encodeChatMessagePrompt(
if (opts?.toolCallId) {
parts.push(encodeString(7, opts.toolCallId));
}
// Accepted live on a tool prompt. Only some models act on it, so the adapter
// also keeps an in-band marker in the text.
if (opts?.isError) parts.push(encodeVarintField(9, 1));
// Assistant message with tool_calls: encode each as a ChatToolCall.
if (opts?.toolCalls && opts.toolCalls.length > 0) {
for (const tc of opts.toolCalls) {
Expand All @@ -257,30 +263,26 @@ function encodeChatMessagePrompt(
const SOURCE_BY_ROLE: Record<string, number> = {
user: 1,
assistant: 2,
// NOTE: do not send source=3 (SYSTEM) directly — the Codeium chat backend
// returns "third-party model provider is experiencing issues" when any
// ChatMessagePrompt has source=SYSTEM. The captured LS upstream traffic
// shows the IDE inlines system context into the *user* prompt (source=1)
// wrapped in <additional_metadata>...</additional_metadata>. We collapse
// role:'system' messages into the next user turn before building the
// proto — see `collapseSystemIntoUser` below.
// Never sent as a prompt source: the leading system text goes in request #2,
// and a later system message is collapsed into the next user turn below.
system: 1,
tool: 4,
};

/**
* Collapse OpenAI-style messages so all `role:'system'` entries are inlined
* into the immediately-following user message, matching the wire format the
* IDE uses. Cognition's chat backend rejects raw role=system entries.
* Collapse `role:'system'` entries that follow the conversation start into the
* immediately-following user message. The leading run of system messages never
* reaches here; it is the request's #2 system prompt. With S0 already sent as #2:
*
* [{system: "S1"}, {system: "S2"}, {user: "U1"}, {assistant: "A1"}, {user: "U2"}]
* [{user: "U1"}, {assistant: "A1"}, {system: "S1"}, {system: "S2"}, {user: "U2"}]
*
* becomes
*
* [{user: "<system>\nS1\nS2\n</system>\nU1"}, {assistant: "A1"}, {user: "U2"}]
* [{user: "U1"}, {assistant: "A1"}, {user: "<system>\nS1\n\nS2\n</system>\nU2"}]
*
* If there's no following user message, the trailing system messages get
* appended as a synthesized user turn.
* appended as a synthesized user turn. A request made only of system messages
* keeps no #2 and comes through here whole, so its prompt list is never empty.
*/
function collapseSystemIntoUser(messages: ChatHistoryItem[]): ChatHistoryItem[] {
const out: ChatHistoryItem[] = [];
Expand Down Expand Up @@ -428,6 +430,8 @@ export interface ChatHistoryItem {
thinking?: string;
signature?: string;
signature_type?: string;
/** For `role: 'tool'` only — the tool failed. Encoded as ChatMessagePrompt #9. */
is_error?: boolean;
}

/**
Expand Down Expand Up @@ -535,8 +539,8 @@ interface BuildArgs {
messages: ChatHistoryItem[];
cascadeId: string;
/**
* GetChatMessageRequest #22. Optional because it is omitted on a first turn;
* the working client only reuses one across a later tool loop.
* GetChatMessageRequest #17 prompt_id. Optional because it is omitted on a
* first turn; the working client only reuses one across a later tool loop.
*/
promptId?: string;
sessionId: string;
Expand Down Expand Up @@ -639,16 +643,20 @@ export function sanitizeToolDescriptionForCognitionForTests(description: string)
return sanitizeToolDescriptionForCognition(description);
}

function encodeToolDef(tool: ToolDef): Buffer {
const rawDesc = sanitizeToolDescriptionForCognition(tool.description ?? '');
const desc =
rawDesc.length > MAX_TOOL_DESC_LEN
? rawDesc.slice(0, MAX_TOOL_DESC_LEN - 24) + '\n…(truncated for cloud)'
: rawDesc;
/** Description as transmitted on the Cognition wire, also used by overflow estimation. */
export function prepareToolDescriptionForCognition(description: string): string {
const rawDesc = sanitizeToolDescriptionForCognition(description);
return rawDesc.length > MAX_TOOL_DESC_LEN
? rawDesc.slice(0, MAX_TOOL_DESC_LEN - 24) + '\n…(truncated for cloud)'
: rawDesc;
}

function encodeToolDef(tool: ToolDef, modelUid: string): Buffer {
const desc = prepareToolDescriptionForCognition(tool.description ?? '');
return Buffer.concat([
encodeString(1, tool.name),
encodeString(2, desc),
encodeString(3, JSON.stringify(tool.parameters ?? {})),
encodeString(3, JSON.stringify(normalizeDevinToolParameters(modelUid, tool.parameters ?? {}))),
]);
}

Expand All @@ -667,9 +675,22 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
cloudChatShape: true,
});

// System messages must be inlined into the user turn (Cognition cloud
// rejects source=3). See `collapseSystemIntoUser` for the format.
const collapsed = collapseSystemIntoUser(args.messages);
// The leading system messages become request #2. Measured live on swe-1-6
// with a ~6.7k-token system prompt: turn 2 read 6688 of 6715 prompt tokens
// from cache in #2, against 7072 of 7097 when the same text was collapsed
// into the first user prompt, so the cache ratio is unchanged and the prompt
// is smaller. The model obeyed an instruction given only in #2.
// A request with only system text keeps it as a user prompt: #2 alone would
// leave the request with no prompt at all.
const firstNonSystem = args.messages.findIndex((m) => m.role !== 'system');
const leadingSystem = firstNonSystem === -1 ? [] : args.messages.slice(0, firstNonSystem);
const systemPrompt = leadingSystem
.map((m) => normalizeContent(m.content)
.filter((p): p is { type: 'text'; text: string } => p.type === 'text')
.map((p) => p.text).join('\n'))
.filter(Boolean)
.join('\n\n');
const collapsed = collapseSystemIntoUser(args.messages.slice(leadingSystem.length));
const promptParts = collapsed.map((m) =>
encodeMessage(
3,
Expand All @@ -686,6 +707,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
thinking: m.role === 'assistant' ? m.thinking : undefined,
signature: m.role === 'assistant' ? m.signature : undefined,
signatureType: m.role === 'assistant' ? m.signature_type : undefined,
isError: m.role === 'tool' ? m.is_error : undefined,
},
),
),
Expand All @@ -694,7 +716,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
const completion = encodeCompletionConfiguration(args.completionOpts ?? {});

const toolParts: Buffer[] = (args.tools ?? []).map((t) =>
encodeMessage(10, encodeToolDef(t)),
encodeMessage(10, encodeToolDef(t, args.modelUid)),
);

// Field layout from mitm capture of the LS:
Expand All @@ -704,15 +726,15 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
// #8 completion_configuration
// #10 tools (repeated ChatToolDefinition)
// #13 prompt_cache_options
// #15 CortexTrajectoryReference
// #16 cascade_id (string)
// #17 prompt_id (string)
// #21 chat_model_uid (string)
// #22 prompt_id (string)
// #22 execution_id (string)
return Buffer.concat([
encodeMessage(1, metadata),
// #2 system_prompt is always written, empty when the caller had none. The
// system turn is separately collapsed into the first user message because
// source=SYSTEM is refused; this field is the one the wire expects here.
encodeString(2, ''),
// #2 system_prompt is always written, empty when the caller had none.
encodeString(2, systemPrompt),
...promptParts,
encodeVarintField(7, args.requestType ?? 5),
encodeMessage(8, completion),
Expand All @@ -724,7 +746,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
// it and records real savings; sending it unconditionally matches both the
// native client and CLIProxyAPIPlus, which places it outside its tools gate.
encodeMessage(13, encodeVarintField(1, PROMPT_CACHE_EPHEMERAL)),
// #15 session model config: { id, turn, 4 }. Present on every verified
// #15 CortexTrajectoryReference: { id, 1, 4 }. Present on every verified
// request.
encodeMessage(15, Buffer.concat([
encodeString(1, crypto.randomUUID()),
Expand All @@ -734,9 +756,9 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
encodeString(16, args.cascadeId),
encodeVarintField(20, 1),
encodeString(21, args.modelUid),
// #22 is deliberately omitted. It is a user-exchange id that only appears
// from the second turn onward and is reused across that turn's tool loop; a
// fresh per-request uuid matches neither shape.
// #17 prompt_id is deliberately omitted. It is a user-exchange id that only
// appears from the second turn onward and is reused across that turn's tool
// loop; a fresh per-request uuid matches neither shape.
]);
}

Expand Down
Loading
Loading