/** * Context Manager — Phase 4 * * Pre-flight context compression to prevent "prompt too long" errors. * 3 layers: trim tool messages, compress structured thinking, aggressive purification. */ import { REGISTRY } from "../config/providerRegistry.ts"; import { getModelContextLimit } from "../../src/lib/modelCapabilities.ts"; import { parseModel } from "./model.ts"; import { jsonLength } from "../utils/jsonSize.ts"; // Default token limits per provider (fallbacks when not in registry) const DEFAULT_LIMITS: Record = { claude: 200000, openai: 128000, gemini: 1000000, codex: 400000, // HyperAgent Claude-family agents (fable/opus/sonnet) — 1M default; was falling // through to 128k and blocking normal agentic tool loops with huge catalogs. hyperagent: 1_000_000, ha: 1_000_000, default: 128000, }; // Environment variable overrides (highest priority) function getEnvOverride(provider: string): number | null { const envKey = `CONTEXT_LENGTH_${provider.toUpperCase().replace(/[^A-Z0-9]/g, "_")}`; const envValue = process.env[envKey]; if (envValue) { const parsed = parseInt(envValue, 10); if (!isNaN(parsed) && parsed > 0) return parsed; } // Global override const globalValue = process.env.CONTEXT_LENGTH_DEFAULT; if (globalValue) { const parsed = parseInt(globalValue, 10); if (!isNaN(parsed) && parsed > 0) return parsed; } return null; } // Reserve tokens override from environment variable function getReserveTokensOverride(): number | null { const envValue = process.env.CONTEXT_RESERVE_TOKENS; if (envValue) { const parsed = parseInt(envValue, 10); if (!isNaN(parsed) && parsed > 0) return parsed; } return null; } // How many of the newest inline images to keep when pruning older ones (#8560). function getKeepLatestImagesOverride(): number | null { const envValue = process.env.CONTEXT_KEEP_LATEST_IMAGES; if (envValue) { const parsed = parseInt(envValue, 10); if (!isNaN(parsed) && parsed >= 0) return parsed; } return null; } const DEFAULT_KEEP_LATEST_IMAGES = 2; const IMAGE_REMOVED_PLACEHOLDER = "[Earlier image removed to fit context window]"; // Rough chars-per-token ratio for quick estimation const CHARS_PER_TOKEN = 4; // Bounded per-image token budget used in place of measuring the raw base64 // payload as text. In line with the owner's PoC (~1052 total for prompt + // 1 image) and litellm's calculate_img_tokens() default-count fast-path — // see #8368 research notes. const IMAGE_TOKEN_ESTIMATE = 1200; // Matches inline base64 data URLs, e.g. "data:image/png;base64,AAAA...". // Deliberately scoped to `data:image/...;base64,` so remote (http/https) // URLs and generic long base64 text strings stay on the text-estimation path. const INLINE_BASE64_IMAGE_RE = /^data:image\/[a-zA-Z0-9.+-]+;base64,/; function isInlineBase64ImageUrl(value: unknown): boolean { return typeof value === "string" && INLINE_BASE64_IMAGE_RE.test(value); } // OpenAI chat.completions: { type: 'image_url', image_url: { url: 'data:...' } | 'data:...' } function matchesOpenAIImageUrlShape(node: Record): boolean { const imageUrl = node.image_url; if (isInlineBase64ImageUrl(imageUrl)) return true; return ( !!imageUrl && typeof imageUrl === "object" && isInlineBase64ImageUrl((imageUrl as Record).url) ); } // AI SDK: { type: 'image', image: 'data:...' } (also covers Responses API's // { type: 'input_image', image_url: 'data:...' } via matchesOpenAIImageUrlShape above). function matchesAiSdkImageShape(node: Record): boolean { return node.type === "image" && isInlineBase64ImageUrl(node.image); } // Claude: { type: 'image', source: { type: 'base64', data: '...' } } function matchesClaudeSourceShape(node: Record): boolean { if (node.type !== "image") return false; const source = node.source; if (!source || typeof source !== "object") return false; const src = source as Record; return src.type === "base64" && typeof src.data === "string"; } // Gemini: { inlineData: { data: '...' } } | { inline_data: { data: '...' } } function matchesGeminiInlineDataShape(node: Record): boolean { const inlineData = node.inlineData ?? node.inline_data; if (!inlineData || typeof inlineData !== "object") return false; return typeof (inlineData as Record).data === "string"; } /** * Detect the 5 documented inline-base64 image content-block shapes (see the * shape-specific matchers above). */ export function isInlineBase64ImageBlock(node: Record): boolean { return ( matchesOpenAIImageUrlShape(node) || matchesAiSdkImageShape(node) || matchesClaudeSourceShape(node) || matchesGeminiInlineDataShape(node) ); } function replaceImageBlockWithPlaceholder(block: Record): Record { // Responses API parts use input_text / input_image — keep that family so restore // does not rewrite a chat-completions-shaped part into a Responses input item. if (block.type === "input_image") { return { type: "input_text", text: IMAGE_REMOVED_PLACEHOLDER }; } if (block.inlineData || block.inline_data) { return { text: IMAGE_REMOVED_PLACEHOLDER }; } return { type: "text", text: IMAGE_REMOVED_PLACEHOLDER }; } /** * Replace oldest inline base64 image blocks with short text placeholders while * keeping the newest `keepLatest` images intact. Vision models still receive * recent screenshots; older ones are dropped so multi-turn Codex/Responses * sessions can fit the concrete input cap (#8560). */ export function pruneOlderInlineImages( messages: Record[], options: { keepLatest?: number; targetTokens?: number } = {} ): { messages: Record[]; pruned: number } { const keepLatest = options.keepLatest ?? getKeepLatestImagesOverride() ?? DEFAULT_KEEP_LATEST_IMAGES; const targetTokens = options.targetTokens; const locations: Array<{ messageIndex: number; contentIndex: number }> = []; for (let messageIndex = 0; messageIndex < messages.length; messageIndex++) { const content = messages[messageIndex]?.content; if (!Array.isArray(content)) continue; for (let contentIndex = 0; contentIndex < content.length; contentIndex++) { const part = content[contentIndex]; if ( part && typeof part === "object" && !Array.isArray(part) && isInlineBase64ImageBlock(part as Record) ) { locations.push({ messageIndex, contentIndex }); } } } if (locations.length <= keepLatest) { return { messages, pruned: 0 }; } const prunable = locations.slice(0, Math.max(0, locations.length - keepLatest)); const next = messages.map((message) => { if (!Array.isArray(message.content)) return message; return { ...message, content: [...message.content] }; }); let pruned = 0; for (const location of prunable) { if (targetTokens != null && estimateTokens(next) <= targetTokens) break; const content = next[location.messageIndex].content as unknown[]; const block = content[location.contentIndex] as Record; content[location.contentIndex] = replaceImageBlockWithPlaceholder(block); pruned += 1; } return { messages: next, pruned }; } /** * Recursively walk a structured node, replacing every recognized inline * base64 image block with a short placeholder (so its bulk is excluded from * the char-count pass below) while accumulating a bounded per-image token * cost. Returns the accumulated image token cost; the caller measures the * placeholder-substituted structure with the normal char/4 heuristic. * * Non-image content (including remote image URLs and generic base64 text) * is left untouched and continues to flow through the text-estimation path. */ function extractImageTokens(node: unknown, seen: Set): { node: unknown; tokens: number } { if (node === null || typeof node !== "object") { return { node, tokens: 0 }; } // Guard against cycles in structured request bodies. if (seen.has(node)) return { node, tokens: 0 }; seen.add(node); if (Array.isArray(node)) { let tokens = 0; const out = node.map((item) => { const record = item && typeof item === "object" && !Array.isArray(item) ? (item as Record) : null; if (record && isInlineBase64ImageBlock(record)) { tokens += IMAGE_TOKEN_ESTIMATE; return { __image_token_estimate__: IMAGE_TOKEN_ESTIMATE }; } const result = extractImageTokens(item, seen); tokens += result.tokens; return result.node; }); return { node: out, tokens }; } const record = node as Record; if (isInlineBase64ImageBlock(record)) { return { node: { __image_token_estimate__: IMAGE_TOKEN_ESTIMATE }, tokens: IMAGE_TOKEN_ESTIMATE, }; } let tokens = 0; const out: Record = {}; for (const [key, value] of Object.entries(record)) { const result = extractImageTokens(value, seen); out[key] = result.node; tokens += result.tokens; } return { node: out, tokens }; } /** * Estimate token count from text length. * * Structured input is first walked for inline base64 image blocks (#8368): * each recognized image block is substituted with a bounded per-image token * budget instead of measuring its base64 payload as raw text, then the * remainder of the structure is measured normally via the char/4 heuristic. */ export function estimateTokens(text: string | object | null | undefined): number { if (!text) return 0; if (typeof text === "string") { return Math.ceil(text.length / CHARS_PER_TOKEN); } const { node, tokens: imageTokens } = extractImageTokens(text, new Set()); // #7847: count the serialized length instead of building the string. Only `.length` was ever // used, and on a multi-megabyte agent body that string is a pure transient allocation. // jsonLength is exact (property-tested against JSON.stringify), so the estimate is unchanged. return Math.ceil(jsonLength(node) / CHARS_PER_TOKEN) + imageTokens; } /** * Get token limit for a provider/model combination * Priority: Env override > models.dev DB > Registry defaultContextLength > DEFAULT_LIMITS */ export function getTokenLimit(provider: string, model: string | null = null): number { return resolveTokenLimit(provider, model).limit; } /** * Resolve a combo target's token limit without crashing when `parseModel(modelStr)` * returns `provider: null` (model id with no `provider/` prefix). * * `ResolvedComboTarget.provider` is populated independently of `modelStr`, so fall * back to it before calling `getTokenLimit` (#8716). */ export function getComboTargetTokenLimit(options: { modelStr?: string | null; provider?: string | null; parsedProvider?: string | null; parsedModel?: string | null; targetProvider?: string | null; }): number { let parsedProvider = options.parsedProvider; let parsedModel = options.parsedModel; if ( (parsedProvider === undefined || parsedModel === undefined) && Object.prototype.hasOwnProperty.call(options, "modelStr") ) { const parsed = parseModel(options.modelStr); if (parsedProvider === undefined) parsedProvider = parsed.provider; if (parsedModel === undefined) parsedModel = parsed.model; } const provider = parsedProvider ?? options.targetProvider ?? options.provider ?? "unknown"; return getTokenLimit(provider, parsedModel ?? null); } /** * Same chain as getTokenLimit, but also reports whether the limit came from * a provider/model-specific source (env override, synced DB, registry, * name heuristic, curated per-provider default) or only from the generic * catch-all default. */ function resolveTokenLimit( provider: string, model: string | null = null ): { limit: number; specific: boolean } { // 1. Check environment variable override first const envOverride = getEnvOverride(provider); if (envOverride) return { limit: envOverride, specific: true }; const lowerModel = (model || "").toLowerCase(); // 2. Check models.dev synced DB for per-model context limit if (model) { const dbLimit = getModelContextLimit(provider, model); if (dbLimit && dbLimit > 0) return { limit: dbLimit, specific: true }; } // 3. Check registry for provider default const registryEntry = REGISTRY[provider]; if (registryEntry?.defaultContextLength) { return { limit: registryEntry.defaultContextLength, specific: true }; } // 4. Check if model name hints at a known limit if (model) { if (lowerModel.includes("claude")) return { limit: DEFAULT_LIMITS.claude, specific: true }; if (lowerModel.includes("gemini")) return { limit: DEFAULT_LIMITS.gemini, specific: true }; if ( lowerModel.includes("gpt") || lowerModel.includes("o1") || lowerModel.includes("o3") || lowerModel.includes("o4") || lowerModel.includes("codex") ) return { limit: DEFAULT_LIMITS.codex, specific: true }; } // 5. Fallback to DEFAULT_LIMITS or default if (DEFAULT_LIMITS[provider]) return { limit: DEFAULT_LIMITS[provider], specific: true }; return { limit: DEFAULT_LIMITS.default, specific: false }; } /** * Resolve the context limit to use for proactive compression of a COMBO * request. * * chatCore always executes with the CONCRETE target's provider/model * (handleSingleModel resolves the target before delegating), so the * executing target's own limit is authoritative. Using min(...allTargets) * here — the previous behavior — compressed at the smallest sibling's * window even when running on the largest target, destructively purging * history long before the real window filled ("agent keeps forgetting"). * * min(...comboTargetLimits) is kept only as a defensive fallback for the * case where the current provider/model resolves no specific limit at all. */ export function resolveComboContextLimit(options: { provider: string; model: string | null; comboTargetLimits: number[]; }): { limit: number; source: "target" | "combo-min" | "fallback" } { const own = resolveTokenLimit(options.provider, options.model ?? null); if (own.specific) { return { limit: own.limit, source: "target" }; } const knownTargets = (options.comboTargetLimits || []).filter( (value) => Number.isFinite(value) && value > 0 ); if (knownTargets.length > 0) { return { limit: Math.min(...knownTargets), source: "combo-min" }; } return { limit: own.limit, source: "fallback" }; } /** * Apply context compression to request body. * Operates in layers of increasing aggressiveness: * * Layer 1: Trim tool_result messages (truncate long outputs) * Layer 1.5: Prune older inline images (keep latest N — #8560) * Layer 2: Compress structured thinking blocks (remove from history, keep last) * Layer 3: Aggressive purification (drop old messages until fitting) * * Callers with OpenAI Responses `input[]` (Codex) must adapt via * `adaptBodyForCompression` before calling and `restore()` after — this helper * is message-centric by design. * * @param {object} body - Request body with messages[] * @param {object} options - { provider?, model?, maxTokens?, reserveTokens?, keepLatestImages? } * @returns {{ body: object, compressed: boolean, stats: object }} */ export function compressContext( body: Record | null | undefined, options: { provider?: string; model?: string; maxTokens?: number; reserveTokens?: number; keepLatestImages?: number; } = {} ) { if (!body || !body.messages || !Array.isArray(body.messages)) { return { body, compressed: false, stats: {} }; } const provider = options.provider || "default"; const maxTokens = options.maxTokens || getTokenLimit(provider, (body.model as string) || options.model || null); const defaultReserveTokens = Math.min(16000, Math.max(256, Math.floor(maxTokens * 0.15))); const reserveTokens = Math.min( options.reserveTokens ?? getReserveTokensOverride() ?? defaultReserveTokens, Math.max(0, maxTokens - 1) ); const targetTokens = Math.max(0, maxTokens - reserveTokens); let messages = [...(body.messages as Record[])]; // #8594: pass the structured messages array directly — estimateTokens walks it for // inline base64 image blocks (#8368) and substitutes a bounded per-image estimate. // JSON.stringify()-ing first forces the char/4 text path and mis-measures a ~500KB // image as ~125k tokens, triggering needless compression / context loss. let currentTokens = estimateTokens(messages); const stats = { original: currentTokens, layers: [] as { name: string; tokens: number }[] }; // Already fits if (currentTokens <= targetTokens) { return { body, compressed: false, stats: { original: currentTokens, final: currentTokens } }; } // Layer 1: Trim tool_result/tool messages messages = trimToolMessages(messages, 2000); // Max 2000 chars per tool result currentTokens = estimateTokens(messages); // #8594: object-path keeps the #8368 image estimate stats.layers.push({ name: "trim_tools", tokens: currentTokens }); if (currentTokens <= targetTokens) { return { body: { ...body, messages }, compressed: true, stats: { ...stats, final: currentTokens }, }; } // Layer 1.5: Drop oldest inline images while keeping the newest ones (#8560). const imagePrune = pruneOlderInlineImages(messages, { keepLatest: options.keepLatestImages, targetTokens, }); if (imagePrune.pruned > 0) { messages = imagePrune.messages; currentTokens = estimateTokens(messages); stats.layers.push({ name: "prune_images", tokens: currentTokens }); if (currentTokens <= targetTokens) { return { body: { ...body, messages }, compressed: true, stats: { ...stats, final: currentTokens }, }; } } // Layer 2: Compress structured thinking blocks (remove from non-last assistant messages) messages = compressThinking(messages); currentTokens = estimateTokens(messages); // #8594: object-path keeps the #8368 image estimate stats.layers.push({ name: "compress_thinking", tokens: currentTokens }); if (currentTokens <= targetTokens) { return { body: { ...body, messages }, compressed: true, stats: { ...stats, final: currentTokens }, }; } // Layer 3: Aggressive purification — drop oldest messages keeping system + last N pairs messages = purifyHistory(messages, targetTokens); currentTokens = estimateTokens(messages); // #8594: object-path keeps the #8368 image estimate stats.layers.push({ name: "purify_history", tokens: currentTokens }); return { body: { ...body, messages }, compressed: true, stats: { ...stats, final: currentTokens }, }; } // ─── Layer 1: Trim Tool Messages ──────────────────────────────────────────── function trimToolMessages(messages: Record[], maxChars: number) { return messages.map((msg) => { if (msg.role === "tool" && typeof msg.content === "string" && msg.content.length > maxChars) { return { ...msg, content: msg.content.slice(0, maxChars) + "\n... [truncated]", }; } // Handle array content (Claude format with tool_result blocks) if (msg.role === "user" && Array.isArray(msg.content)) { return { ...msg, content: msg.content.map((block) => { if ( block.type === "tool_result" && typeof block.content === "string" && block.content.length > maxChars ) { return { ...block, content: block.content.slice(0, maxChars) + "\n... [truncated]" }; } return block; }), }; } return msg; }); } // ─── Layer 2: Compress Structured Thinking Blocks ─────────────────────────── function compressThinking(messages: Record[]) { // Find last assistant message index let lastAssistantIdx = -1; for (let i = messages.length - 1; i >= 0; i--) { if (messages[i].role === "assistant") { lastAssistantIdx = i; break; } } return messages.map((msg, i) => { if (msg.role !== "assistant") return msg; if (i === lastAssistantIdx) return msg; // Keep thinking in last assistant msg // Remove thinking blocks from content array if (Array.isArray(msg.content)) { const filtered = msg.content.filter((block) => block.type !== "thinking"); if (filtered.length === 0) { return { ...msg, content: "[thinking compressed]" }; } return { ...msg, content: filtered }; } return msg; }); } // ─── Layer 3: Aggressive Purification ─────────────────────────────────────── function purifyHistory(messages: Record[], targetTokens: number) { // Keep system message(s) and the last N message pairs const system = messages.filter((m) => m.role === "system" || m.role === "developer"); const nonSystem = messages.filter((m) => m.role !== "system" && m.role !== "developer"); // Binary search for how many messages to keep from the end let keep = nonSystem.length; while (keep > 2) { let candidate = [...system, ...nonSystem.slice(-keep)]; candidate = fixToolPairs(candidate); candidate = fixToolAdjacency(candidate); // Re-run pair fix: fixToolAdjacency may have stripped tool_use blocks, leaving // orphan tool_results that Claude rejects ("tool_result without preceding tool_use"). candidate = fixToolPairs(candidate); candidate = stripTrailingAssistantOrphanToolUse(candidate); // #8594: measure the candidate structure directly so image-bearing turns are not // over-counted and pruned during the binary search. const tokens = estimateTokens(candidate); if (tokens <= targetTokens) break; keep = Math.max(2, Math.floor(keep * 0.7)); // Drop 30% each iteration } let result = [...system, ...nonSystem.slice(-keep)]; result = fixToolPairs(result); result = fixToolAdjacency(result); // Re-run pair fix to drop any tool_result whose matching tool_use was removed by // fixToolAdjacency (discussion #2410 — orphan tool_result -> upstream 400). result = fixToolPairs(result); result = stripTrailingAssistantOrphanToolUse(result); // Add summary of dropped messages if (keep < nonSystem.length) { const dropped = nonSystem.length - keep; result.splice(system.length, 0, { role: "system", content: `[Context compressed: ${dropped} earlier messages removed to fit context window]`, }); } return result; } /** * Remove orphaned tool_result messages whose preceding tool_use was dropped. * Also removes orphaned tool_use messages without a corresponding tool_result. * * When purifyHistory() drops oldest messages, it can split tool_use/tool_result * pairs — keeping the tool_result but dropping the tool_use that initiated it. * This causes upstream providers to reject the request with errors like: * - Claude: "tool_result message must be preceded by a tool_use message" * - OpenAI: "Invalid message format" * - Gemini: "Function response without function call" */ export function fixToolPairs(messages: Record[]) { // Pass 1: Collect all tool_result IDs from user/tool messages const toolResultIds = new Set(); for (const msg of messages) { if (msg.role === "tool" && msg.tool_call_id) { toolResultIds.add(msg.tool_call_id); } if (msg.role === "user" && Array.isArray(msg.content)) { for (const block of msg.content) { if (block.type === "tool_result" && block.tool_use_id) { toolResultIds.add(block.tool_use_id); } } } } // Pass 2: Filter assistant messages to remove tool_use without tool_result // (Exception: keep tool_use if the assistant message is the last message) const isLastMessage = (idx: number) => idx === messages.length - 1; const filteredMessages = messages.map((msg, idx) => { if (msg.role === "assistant" && !isLastMessage(idx)) { let modified = false; const newMsg = { ...msg }; if (Array.isArray(newMsg.tool_calls)) { const filteredToolCalls = newMsg.tool_calls.filter( (tc: Record) => !tc.id || toolResultIds.has(tc.id) ); if (filteredToolCalls.length !== newMsg.tool_calls.length) { newMsg.tool_calls = filteredToolCalls; modified = true; } } if (Array.isArray(newMsg.content)) { const filteredContent = newMsg.content.filter( (block: Record) => block.type !== "tool_use" || !block.id || toolResultIds.has(block.id) ); if (filteredContent.length !== newMsg.content.length) { newMsg.content = filteredContent; modified = true; } } return modified ? newMsg : msg; } return msg; }); // Pass 3: Collect all remaining tool_use IDs from assistant messages const toolCallIds = new Set(); for (const msg of filteredMessages) { if (msg.role === "assistant") { if (Array.isArray(msg.tool_calls)) { for (const tc of msg.tool_calls) { if (tc.id) toolCallIds.add(tc.id); } } if (Array.isArray(msg.content)) { for (const block of msg.content) { if (block.type === "tool_use" && block.id) { toolCallIds.add(block.id); } } } } } // Pass 4: Filter user/tool messages to remove tool_result without tool_use return filteredMessages .map((msg) => { if (msg.role === "tool" && msg.tool_call_id) { if (!toolCallIds.has(msg.tool_call_id)) return null; } if (msg.role === "user" && Array.isArray(msg.content)) { const filteredContent = msg.content.filter( (block: Record) => block.type !== "tool_result" || !block.tool_use_id || toolCallIds.has(block.tool_use_id) ); if (filteredContent.length !== msg.content.length) { if (filteredContent.length === 0) return null; return { ...msg, content: filteredContent }; } } // Drop assistant messages if their content AND tool_calls became empty if (msg.role === "assistant") { const hasContent = typeof msg.content === "string" ? msg.content.trim().length > 0 : Array.isArray(msg.content) && msg.content.length > 0; const hasToolCalls = Array.isArray(msg.tool_calls) && msg.tool_calls.length > 0; if (!hasContent && !hasToolCalls) { return null; } } return msg; }) .filter(Boolean) as Record[]; } /** * Adjacency guard: Claude requires `tool_result` in the IMMEDIATELY NEXT * message after `tool_use`, not just somewhere later in the array. * * `fixToolPairs` checks global ID presence but not adjacency. This function * runs after `fixToolPairs` and removes `tool_use` blocks from assistant * messages where the next message does not contain a matching `tool_result`. */ export function fixToolAdjacency(messages: Record[]): Record[] { if (messages.length <= 1) return messages; const result: Record[] = []; for (let i = 0; i < messages.length; i++) { const msg = messages[i]; const nextMsg = messages[i + 1]; if (msg.role !== "assistant" || !nextMsg) { result.push(msg); continue; } // Collect tool_result IDs from the NEXT message only const nextToolResultIds = new Set(); if (nextMsg.role === "tool" && nextMsg.tool_call_id) { nextToolResultIds.add(String(nextMsg.tool_call_id)); } if (nextMsg.role === "user" && Array.isArray(nextMsg.content)) { for (const block of nextMsg.content as Record[]) { if (block.type === "tool_result" && block.tool_use_id) { nextToolResultIds.add(String(block.tool_use_id)); } } } let modified = false; const newMsg: Record = { ...msg }; // Filter tool_use blocks in content array (Claude format) if (Array.isArray(newMsg.content)) { const filteredContent = (newMsg.content as Record[]).filter( (block) => block.type !== "tool_use" || !block.id || nextToolResultIds.has(String(block.id)) ); if (filteredContent.length !== (newMsg.content as unknown[]).length) { newMsg.content = filteredContent; modified = true; } } // Filter tool_calls array (OpenAI format) — independently of content if (Array.isArray(newMsg.tool_calls)) { const filteredToolCalls = (newMsg.tool_calls as Record[]).filter( (tc: Record) => !tc.id || nextToolResultIds.has(String(tc.id)) ); if (filteredToolCalls.length !== (newMsg.tool_calls as unknown[]).length) { newMsg.tool_calls = filteredToolCalls; modified = true; } } if (modified) { // Drop assistant message if it became empty const hasContent = typeof newMsg.content === "string" ? (newMsg.content as string).trim().length > 0 : Array.isArray(newMsg.content) && (newMsg.content as unknown[]).length > 0; const hasToolCalls = Array.isArray(newMsg.tool_calls) && newMsg.tool_calls.length > 0; if (!hasContent && !hasToolCalls) continue; result.push(newMsg); } else { result.push(msg); } } return result; } /** * Upstream-send guard: after `fixToolPairs`, strip a trailing assistant * message whose only/remaining content is an orphan `tool_use` block. * * `fixToolPairs` intentionally preserves a final-message `tool_use` because * during context pruning the client is still waiting on the matching * `tool_result` — dropping it there would lose state. But on the * upstream-send path the request body must end on a user turn; a trailing * `assistant(tool_use)` triggers the same Anthropic 400 the guard is * trying to prevent: * messages.N: `tool_use` ids were found without `tool_result` blocks * immediately after: toolu_... * * Behavior: * - If the last message is `assistant` and contains any `tool_use` block, * those blocks are removed. * - If removal leaves the message with no content / tool_calls at all, the * message itself is dropped. * - Idempotent on clean histories (trailing user, trailing assistant with * only text/thinking, etc.). */ export function stripTrailingAssistantOrphanToolUse( messages: Record[] ): Record[] { if (!Array.isArray(messages) || messages.length === 0) return messages; const lastIdx = messages.length - 1; const last = messages[lastIdx]; if (!last || last.role !== "assistant") return messages; let modified = false; const newLast: Record = { ...last }; if (Array.isArray(newLast.tool_calls)) { const filteredCalls = (newLast.tool_calls as Record[]).filter( () => false // remove all trailing tool_calls (none can be paired by definition) ); if (filteredCalls.length !== (newLast.tool_calls as unknown[]).length) { newLast.tool_calls = filteredCalls; modified = true; } } if (Array.isArray(newLast.content)) { const filteredContent = (newLast.content as Record[]).filter( (block) => block.type !== "tool_use" ); if (filteredContent.length !== (newLast.content as unknown[]).length) { newLast.content = filteredContent; modified = true; } } if (!modified) return messages; // If the last message is now empty, drop it. const hasContent = typeof newLast.content === "string" ? (newLast.content as string).trim().length > 0 : Array.isArray(newLast.content) && (newLast.content as unknown[]).length > 0; const hasToolCalls = Array.isArray(newLast.tool_calls) && (newLast.tool_calls as unknown[]).length > 0; const result = messages.slice(0, lastIdx); if (hasContent || hasToolCalls) result.push(newLast); return result; } /** * Providers that strictly require the last message to be `user` or `tool`. * A trailing `assistant` message with plain text content (no tool_use) is * valid for Anthropic/OpenAI (signals "continue from here") but rejected by * Mistral with: "Expected last role User or Tool … but got assistant" (#3396). */ const PROVIDERS_REQUIRING_USER_LAST_MESSAGE = new Set(["mistral"]); /** * Strip a trailing `assistant` message that contains ONLY plain text (no * `tool_use` / `tool_calls`) for providers that mandate user-last format. * * Call this AFTER `stripTrailingAssistantOrphanToolUse` on the upstream-send * path so `tool_use` orphans are already removed before this check runs. */ export function stripTrailingAssistantForProvider( messages: Record[], provider: string ): Record[] { if (!PROVIDERS_REQUIRING_USER_LAST_MESSAGE.has(provider)) return messages; if (!Array.isArray(messages) || messages.length === 0) return messages; const last = messages[messages.length - 1]; if (!last || last.role !== "assistant") return messages; // Only strip when the message has NO tool_use / tool_calls (those are // handled by stripTrailingAssistantOrphanToolUse upstream of this call). const hasToolUse = Array.isArray(last.content) && (last.content as Record[]).some((b) => b.type === "tool_use"); const hasToolCalls = Array.isArray(last.tool_calls) && (last.tool_calls as unknown[]).length > 0; if (hasToolUse || hasToolCalls) return messages; return messages.slice(0, messages.length - 1); }