mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-31 12:22:14 +03:00
* fix(cli-tools): guard modelId type before calling indexOf E2E shakedown v3.8.0: cli-tools quebrava com TypeError quando dynamicModels continha entradas sem .id (objeto retornado diretamente em vez de string). * fix(offline): avoid SSR/CSR hydration mismatch on navigator.onLine Replace useState+lazy-initializer with useSyncExternalStore so the server snapshot (() => false) and client snapshot (() => navigator.onLine) are declared separately. React hydrates with the server value and switches to the real online status client-side without a mismatch. * chore(i18n): add missing en.json keys for translator, cli-tools, memory, onboarding Adds 58 missing keys identified by the new dashboard audit script: - cliTools: 18 custom CLI builder keys (CustomCliCard) - translator: 24 keys covering stream transformer, live monitor, test bench - memory: 12 health/pagination/dialog keys - onboarding.tier: 8 keys for the tier tour walkthrough Also adds scripts/i18n/audit-dashboard-pages.mjs which scans all dashboard pages, reports t() calls referencing missing en.json keys, and flags candidate hardcoded JSX/attribute strings. * chore(i18n): replace hardcoded UI text with t() calls across dashboard (round 1) Subagents refactored 8 high-impact dashboard pages, replacing 81 of the 407 hardcoded English/PT strings flagged by the audit with proper useTranslations() lookups. Added 73 corresponding keys to en.json across the home, apiManager, providers, settings, and usage namespaces. Pages affected: - BudgetTab (27 → 0) - HomePageClient (2 → 0) - RoutingTab (25 → 7) - ResilienceTab (38 → 18) - SystemStorageTab (42 → 21) - providers/[id] (17 → 15) - ApiManagerPageClient (14 → 13) - OneproxyTab (13 → 10) Also adds two helper scripts: - scripts/i18n/extract-keys-from-diff.mjs — extracts new keys from git diff - scripts/i18n/merge-keys.mjs — merges a pending-keys JSON into en.json Remaining hardcoded strings will be addressed in follow-up rounds. * chore(i18n): replace hardcoded UI text with t() calls across dashboard (round 2) Continues round 1 (commit8d34f4c65). Round-2 subagents refactored additional dashboard pages, replacing 77 more hardcoded strings with useTranslations() lookups. Added 79 corresponding keys to en.json across the a2aDashboard, agents, analytics, apiManager, cliTools, common, and settings namespaces. Pages affected: - a2a/page (new useTranslations + 6 keys) - agent-skills/page (new useTranslations + 9 keys) - AutoRoutingAnalyticsTab (new useTranslations + 6 keys) - AppearanceTab (8 → 6 remaining) - OneproxyTab (10 → 0) - ResilienceTab (18 → 0 missing key) - RoutingTab (7 → 0 missing key) - VisionBridgeSettingsTab (new useTranslations + 6 keys) - CopilotToolCard (7 → 0 missing key) - ApiManagerPageClient (13 → 0 missing key) - gamification/admin (new useTranslations + 7 keys) Hardcoded total: 326 → 249. Real missing keys: 0 (the 6 still flagged are false positives in exampleTemplates.tsx where t is passed as a parameter — keys exist at translator.templatePayloads.*). * chore(i18n): replace hardcoded UI text with t() calls across dashboard (round 3) Round-3 subagents and manual edits refactored 9 more dashboard pages (plus 2 small extras), replacing ~80 hardcoded strings with useTranslations() lookups. Added 79 corresponding keys to en.json across analytics, cloudAgents, combos, common, health, settings, and usage namespaces. Pages affected: - analytics/ComboHealthTab (new useTranslations + 15 keys) - analytics/CompressionAnalyticsTab (new useTranslations + 11 keys) - settings/SystemStorageTab (21 → 0 missing key) - tokens/page (new useTranslations + 13 keys) - usage/BudgetTab (9 missing fixed) - health/page (manual: 6 keys) - cloud-agents/page (manual: 3 keys) - combos/page (manual: 1 key) Hardcoded total: 249 → 164. Real missing keys: 0 (6 remaining are exampleTemplates.tsx false positives). Also adds scripts/i18n/build-pending-from-missing.mjs which reads _audit.json and locates English values from HEAD to rebuild _pending-keys.json after race-condition resets between subagent edits. * chore(i18n): localize remaining dashboard settings labels Replace hardcoded labels in compression and resilience settings with translation lookups to continue the dashboard i18n cleanup. Add the v3.8.0 dashboard shakedown runbook to document the manual smoke-test process and known dev environment pitfalls. * chore(i18n): replace hardcoded UI text with t() calls across dashboard (round 4) Round-4 subagent + manual key-resolution refactored remaining strings in 3 high-traffic settings/API tabs, plus extracted English values for keys that were already added as t() calls but lost during the previous en.json race-condition resets. Pages affected: - api-manager/ApiManagerPageClient (7 → 0 missing key) - settings/CompressionSettingsTab (8 → 0 missing key) - settings/MemorySkillsTab (8 → 0 missing key) - settings/ResilienceTab (4 more keys recovered) Hardcoded total: 164 → 140. Real missing keys: 0 (6 remaining are the exampleTemplates.tsx false positives — t passed as parameter). * chore(i18n): replace hardcoded UI text with t() calls across dashboard (round 5) Round-5 agent began processing the remaining smaller dashboard files. Added 5 more keys to en.json for providers/[id]/page.tsx OAuth flow labels and the cross-OS auto-detection hint. Pages affected: - providers/[id]/page.tsx (5 keys) Hardcoded total: 140 → 136. Real missing keys: 0. * chore(i18n): resolve last 2 missing providers/[id] keys Adds providerDetailMyClaudeAccountPlaceholder and providerDetailPathAutoDetected — the final user-visible labels in the providers/[id] page that the round-5 subagent rewrote to t() calls without yet adding to en.json. Real missing keys: 0 (6 remaining are exampleTemplates.tsx false positives — t is passed as a parameter so the audit cannot resolve the namespace; keys do exist at translator.templatePayloads.*). * chore(i18n): replace hardcoded UI text with t() calls across dashboard (round 6 — 10 parallel agents) Round-6 dispatched 10 parallel subagents covering all 57 remaining dashboard files. Each agent worked on a disjoint file set to avoid en.json race conditions. Added ~60 new i18n keys across 9 namespaces covering small UI labels, table headers, search placeholders, and empty-state messages. Major changes: - analytics: SearchAnalyticsTab, ProviderUtilizationTab, DiversityScoreCard, CompressionAnalyticsTab (new useTranslations + keys) - batch: BatchDetailModal, BatchListTab, FileDetailModal, FilesListTab (new useTranslations + keys) - settings: CliproxyapiSettingsTab, PayloadRulesTab, ModelCooldownsCard, AppearanceTab, PricingTab (mostly new useTranslations) - endpoint: TokenSaverCard, ApiEndpointsTab, EndpointPageClient - cache: CachePerformance, IdempotencyLayer, ReasoningCacheTab, MediaPageClient, page - combos: IntelligentComboPanel, page - playground: ChatPlayground, SearchPlayground - providers: ProviderCard - onboarding: TierFlowDiagram - changelog: ChangelogViewer - home: ProviderTopology, TierCoverageWidget, BootstrapBanner, BadgeToast - usage: BudgetTab, BudgetTelemetryCards, QuotaTable - quotaShare: QuotaSharePageClient - profile: page - leaderboard: page - skills: page Hardcoded total: 131 → 60. Real missing keys: 0 plus 1 false-positive for combos.modePack (lookup via prop-passed t). * chore(i18n): finalize round-6 keys for batch/cache/endpoint/usage Adds the remaining keys produced by parallel agents A4, A6, A8, A9: - common: batch-related labels (BatchDetailModal, BatchListTab, FileDetailModal, FilesListTab, page) + profile/leaderboard - cache: hit rate, latency, retry, avg chars - endpoint: token saver, API endpoints, copy URL, cloud/local labels - usage: noSpend, activeSessions, quotaAlerts, budget timing - skills: install/marketplace/filter - proxyRegistry/quotaShare/mcpDashboard: misc labels Hardcoded total: 60 → 48. Real missing keys: 0 (modePack remaining is a false positive — combos.modePack exists but the audit can't resolve it since IntelligentComboPanel receives t as a prop). * fix(playground): dedupe filteredModels to avoid duplicate React key warning The /v1/models endpoint can return the same model id twice (e.g., when a model is listed by both an alias and its canonical provider), which made the <Select> emit two <option> elements with the same key — triggering "Encountered two children with the same key, codex/gpt-5.5". Replace the chained filter + map with a single pass that skips ids already added. * fix(playground): guard against non-string model ids before .split/.startsWith The /v1/models endpoint can include synthetic entries (combos, locals, in-progress imports) with a null/undefined id. The playground used to call m.id.split("/") in the provider-discovery loop, which threw on the first non-string entry; the surrounding .catch(() => {}) silently swallowed the error, so the provider/model/account dropdowns ended up empty even though /v1/models returned thousands of valid entries. - Skip entries without a string id before split/startsWith. - Log the rejection in the .catch handler so future regressions are visible in DevTools instead of silently emptying the UI. * fix(playground): guard ChatPlayground filteredModels for non-string ids Same root cause as commit49fe356b9: ChatPlayground filtered models with m.id.startsWith(...) which crashed on null/undefined ids returned by /v1/models (synthetic combo entries). Apply the same defensive guard and dedupe used in the parent page. * fix(claude): drop orphan tool_result after fixToolAdjacency strip (discussion #2410) Discussion #2410 reports Claude returning 400 for sequences like: assistant: tool_use(id=X) user: <plain text> ← breaks adjacency user: tool_result(id=X) The previous round added `fixToolAdjacency` (commit44d9abac9) which correctly strips the orphan tool_use from the assistant message. But that left the now-unmatched tool_result intact, so the upstream rejected the request with: messages.N.content.M: unexpected `tool_use_id` found in `tool_result` blocks: X. Each tool_result block must have a corresponding tool_use block in the previous message. Fix: after running `fixToolAdjacency`, re-run `fixToolPairs` to drop the orphaned tool_result blocks. All three call sites updated: - contextManager.purifyHistory (both inside the binary-search loop and the final pass) - BaseExecutor message-prep (Claude path) - claudeCodeCompatible request signer Also tightens an unrelated dynamic-key access in readNestedString (claudeCodeCompatible) to satisfy the prototype- pollution scanner triggered by the post-tool semgrep hook. * fix(mitm): point runtime manager re-export to js entrypoint Use the emitted `.js` path for the runtime manager re-export so dynamic runtime loading resolves correctly outside the Turbopack alias handling. * docs: add AgentRouter setup guide (#2422) Integrated into release/v3.8.0 — AgentRouter setup guide docs. * feat: add new feature on combos - falloverBeforeRetry (#2417) Integrated into release/v3.8.0 — falloverBeforeRetry for per-model quota skipping in combos. * feat(batch): implement 10 feature requests harvested (#2414) Integrated into release/v3.8.0 — batch of 10 feature requests: llama.cpp local provider, upstream error exposure, Termux detection, providers rotate CLI, t3.chat web skeleton, Zed Docker integration, Kiro multi-account OAuth isolation, auto-combo cost blending, auto-combo context filter, combo provider-level exhaustion tracking (#1731). Conflicts with #2417 (falloverBeforeRetry) resolved. * fix(gamification): resolve SQL bug, auth gap, pagination, and anomaly scoring (#2421) Integrated into release/v3.8.0 — 6 critical gamification bug fixes: SQL SELECT in checkActionCountBadges, federation auth enforcement, leaderboard pagination offset, real z-score computation, addXp level calculation, and barrel index.ts * docs(changelog): add post-release entries for #2414 #2417 #2421 #2422 - feat(batch): T3-Chat-Web executor, exhaustedProviders set (#1731), Zed Docker - feat(combos): falloverBeforeRetry + setTry loop (#2417 — @hartmark) - fix(gamification): SQL SELECT bug, federation auth, pagination, z-score (#2421 — @oyi77) - docs: AgentRouter setup guide (#2422 — @leninejunior) * fix(security): resolve CodeQL random/password-hash alerts and sync docs & tests --------- Co-authored-by: diegosouzapw <diego.souza.pw@gmail.com> Co-authored-by: Lenine Júnior <lenine@engrene.com.br> Co-authored-by: Markus Hartung <mail@hartmark.se> Co-authored-by: Paijo <14921983+oyi77@users.noreply.github.com>
558 lines
20 KiB
TypeScript
558 lines
20 KiB
TypeScript
/**
|
|
* Context Manager — Phase 4
|
|
*
|
|
* Pre-flight context compression to prevent "prompt too long" errors.
|
|
* 3 layers: trim tool messages, compress thinking, aggressive purification.
|
|
*/
|
|
|
|
import { REGISTRY } from "../config/providerRegistry.ts";
|
|
import { getModelContextLimit } from "../../src/lib/modelCapabilities.ts";
|
|
|
|
// Default token limits per provider (fallbacks when not in registry)
|
|
const DEFAULT_LIMITS: Record<string, number> = {
|
|
claude: 200000,
|
|
openai: 128000,
|
|
gemini: 1000000,
|
|
codex: 400000,
|
|
default: 128000,
|
|
};
|
|
|
|
// Environment variable overrides (highest priority)
|
|
function getEnvOverride(provider: string): number | null {
|
|
const envKey = `CONTEXT_LENGTH_${provider.toUpperCase().replace(/[^A-Z0-9]/g, "_")}`;
|
|
const envValue = process.env[envKey];
|
|
if (envValue) {
|
|
const parsed = parseInt(envValue, 10);
|
|
if (!isNaN(parsed) && parsed > 0) return parsed;
|
|
}
|
|
// Global override
|
|
const globalValue = process.env.CONTEXT_LENGTH_DEFAULT;
|
|
if (globalValue) {
|
|
const parsed = parseInt(globalValue, 10);
|
|
if (!isNaN(parsed) && parsed > 0) return parsed;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
// Reserve tokens override from environment variable
|
|
function getReserveTokensOverride(): number | null {
|
|
const envValue = process.env.CONTEXT_RESERVE_TOKENS;
|
|
if (envValue) {
|
|
const parsed = parseInt(envValue, 10);
|
|
if (!isNaN(parsed) && parsed > 0) return parsed;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
// Rough chars-per-token ratio for quick estimation
|
|
const CHARS_PER_TOKEN = 4;
|
|
|
|
/**
|
|
* Estimate token count from text length
|
|
*/
|
|
export function estimateTokens(text: string | object | null | undefined): number {
|
|
if (!text) return 0;
|
|
const str = typeof text === "string" ? text : JSON.stringify(text);
|
|
return Math.ceil(str.length / CHARS_PER_TOKEN);
|
|
}
|
|
|
|
/**
|
|
* Get token limit for a provider/model combination
|
|
* Priority: Env override > models.dev DB > Registry defaultContextLength > DEFAULT_LIMITS
|
|
*/
|
|
export function getTokenLimit(provider: string, model: string | null = null): number {
|
|
// 1. Check environment variable override first
|
|
const envOverride = getEnvOverride(provider);
|
|
if (envOverride) return envOverride;
|
|
|
|
// 2. Check models.dev synced DB for per-model context limit
|
|
if (model) {
|
|
const dbLimit = getModelContextLimit(provider, model);
|
|
if (dbLimit && dbLimit > 0) return dbLimit;
|
|
}
|
|
|
|
// 3. Check registry for provider default
|
|
const registryEntry = REGISTRY[provider];
|
|
if (registryEntry?.defaultContextLength) {
|
|
return registryEntry.defaultContextLength;
|
|
}
|
|
|
|
// 4. Check if model name hints at a known limit
|
|
if (model) {
|
|
const lower = model.toLowerCase();
|
|
if (lower.includes("claude")) return DEFAULT_LIMITS.claude;
|
|
if (lower.includes("gemini")) return DEFAULT_LIMITS.gemini;
|
|
if (
|
|
lower.includes("gpt") ||
|
|
lower.includes("o1") ||
|
|
lower.includes("o3") ||
|
|
lower.includes("o4") ||
|
|
lower.includes("codex")
|
|
)
|
|
return DEFAULT_LIMITS.codex;
|
|
}
|
|
|
|
// 5. Fallback to DEFAULT_LIMITS or default
|
|
return DEFAULT_LIMITS[provider] || DEFAULT_LIMITS.default;
|
|
}
|
|
|
|
/**
|
|
* Apply context compression to request body.
|
|
* Operates in 3 layers of increasing aggressiveness:
|
|
*
|
|
* Layer 1: Trim tool_result messages (truncate long outputs)
|
|
* Layer 2: Compress thinking blocks (remove from history, keep last)
|
|
* Layer 3: Aggressive purification (drop old messages until fitting)
|
|
*
|
|
* @param {object} body - Request body with messages[]
|
|
* @param {object} options - { provider?, model?, maxTokens?, reserveTokens? }
|
|
* @returns {{ body: object, compressed: boolean, stats: object }}
|
|
*/
|
|
export function compressContext(
|
|
body: Record<string, unknown>,
|
|
options: { provider?: string; model?: string; maxTokens?: number; reserveTokens?: number } = {}
|
|
) {
|
|
if (!body || !body.messages || !Array.isArray(body.messages)) {
|
|
return { body, compressed: false, stats: {} };
|
|
}
|
|
|
|
const provider = options.provider || "default";
|
|
const maxTokens =
|
|
options.maxTokens || getTokenLimit(provider, (body.model as string) || options.model || null);
|
|
const defaultReserveTokens = Math.min(16000, Math.max(256, Math.floor(maxTokens * 0.15)));
|
|
const reserveTokens = Math.min(
|
|
options.reserveTokens ?? getReserveTokensOverride() ?? defaultReserveTokens,
|
|
Math.max(0, maxTokens - 1)
|
|
);
|
|
const targetTokens = Math.max(0, maxTokens - reserveTokens);
|
|
|
|
let messages = [...body.messages];
|
|
let currentTokens = estimateTokens(JSON.stringify(messages));
|
|
const stats = { original: currentTokens, layers: [] as { name: string; tokens: number }[] };
|
|
|
|
// Already fits
|
|
if (currentTokens <= targetTokens) {
|
|
return { body, compressed: false, stats: { original: currentTokens, final: currentTokens } };
|
|
}
|
|
|
|
// Layer 1: Trim tool_result/tool messages
|
|
messages = trimToolMessages(messages, 2000); // Max 2000 chars per tool result
|
|
currentTokens = estimateTokens(JSON.stringify(messages));
|
|
stats.layers.push({ name: "trim_tools", tokens: currentTokens });
|
|
|
|
if (currentTokens <= targetTokens) {
|
|
return {
|
|
body: { ...body, messages },
|
|
compressed: true,
|
|
stats: { ...stats, final: currentTokens },
|
|
};
|
|
}
|
|
|
|
// Layer 2: Compress thinking blocks (remove from non-last assistant messages)
|
|
messages = compressThinking(messages);
|
|
currentTokens = estimateTokens(JSON.stringify(messages));
|
|
stats.layers.push({ name: "compress_thinking", tokens: currentTokens });
|
|
|
|
if (currentTokens <= targetTokens) {
|
|
return {
|
|
body: { ...body, messages },
|
|
compressed: true,
|
|
stats: { ...stats, final: currentTokens },
|
|
};
|
|
}
|
|
|
|
// Layer 3: Aggressive purification — drop oldest messages keeping system + last N pairs
|
|
messages = purifyHistory(messages, targetTokens);
|
|
currentTokens = estimateTokens(JSON.stringify(messages));
|
|
stats.layers.push({ name: "purify_history", tokens: currentTokens });
|
|
|
|
return {
|
|
body: { ...body, messages },
|
|
compressed: true,
|
|
stats: { ...stats, final: currentTokens },
|
|
};
|
|
}
|
|
|
|
// ─── Layer 1: Trim Tool Messages ────────────────────────────────────────────
|
|
|
|
function trimToolMessages(messages: Record<string, unknown>[], maxChars: number) {
|
|
return messages.map((msg) => {
|
|
if (msg.role === "tool" && typeof msg.content === "string" && msg.content.length > maxChars) {
|
|
return {
|
|
...msg,
|
|
content: msg.content.slice(0, maxChars) + "\n... [truncated]",
|
|
};
|
|
}
|
|
// Handle array content (Claude format with tool_result blocks)
|
|
if (msg.role === "user" && Array.isArray(msg.content)) {
|
|
return {
|
|
...msg,
|
|
content: msg.content.map((block) => {
|
|
if (
|
|
block.type === "tool_result" &&
|
|
typeof block.content === "string" &&
|
|
block.content.length > maxChars
|
|
) {
|
|
return { ...block, content: block.content.slice(0, maxChars) + "\n... [truncated]" };
|
|
}
|
|
return block;
|
|
}),
|
|
};
|
|
}
|
|
return msg;
|
|
});
|
|
}
|
|
|
|
// ─── Layer 2: Compress Thinking Blocks ──────────────────────────────────────
|
|
|
|
function compressThinking(messages: Record<string, unknown>[]) {
|
|
// Find last assistant message index
|
|
let lastAssistantIdx = -1;
|
|
for (let i = messages.length - 1; i >= 0; i--) {
|
|
if (messages[i].role === "assistant") {
|
|
lastAssistantIdx = i;
|
|
break;
|
|
}
|
|
}
|
|
|
|
return messages.map((msg, i) => {
|
|
if (msg.role !== "assistant") return msg;
|
|
if (i === lastAssistantIdx) return msg; // Keep thinking in last assistant msg
|
|
|
|
// Remove thinking blocks from content array
|
|
if (Array.isArray(msg.content)) {
|
|
const filtered = msg.content.filter((block) => block.type !== "thinking");
|
|
if (filtered.length === 0) {
|
|
return { ...msg, content: "[thinking compressed]" };
|
|
}
|
|
return { ...msg, content: filtered };
|
|
}
|
|
|
|
// Remove thinking XML tags from string content
|
|
if (typeof msg.content === "string") {
|
|
let cleaned = msg.content;
|
|
for (const [start, end] of [
|
|
["<thinking>", "</thinking>"],
|
|
["<antThinking>", "</antThinking>"],
|
|
]) {
|
|
while (true) {
|
|
const s = cleaned.indexOf(start);
|
|
if (s === -1) break;
|
|
const e = cleaned.indexOf(end, s + start.length);
|
|
if (e === -1) {
|
|
cleaned = cleaned.slice(0, s);
|
|
break;
|
|
}
|
|
cleaned = cleaned.slice(0, s) + cleaned.slice(e + end.length);
|
|
}
|
|
}
|
|
cleaned = cleaned.trim();
|
|
return { ...msg, content: cleaned || "[thinking compressed]" };
|
|
}
|
|
|
|
return msg;
|
|
});
|
|
}
|
|
|
|
// ─── Layer 3: Aggressive Purification ───────────────────────────────────────
|
|
|
|
function purifyHistory(messages: Record<string, unknown>[], targetTokens: number) {
|
|
// Keep system message(s) and the last N message pairs
|
|
const system = messages.filter((m) => m.role === "system" || m.role === "developer");
|
|
const nonSystem = messages.filter((m) => m.role !== "system" && m.role !== "developer");
|
|
|
|
// Binary search for how many messages to keep from the end
|
|
let keep = nonSystem.length;
|
|
while (keep > 2) {
|
|
let candidate = [...system, ...nonSystem.slice(-keep)];
|
|
candidate = fixToolPairs(candidate);
|
|
candidate = fixToolAdjacency(candidate);
|
|
// Re-run pair fix: fixToolAdjacency may have stripped tool_use blocks, leaving
|
|
// orphan tool_results that Claude rejects ("tool_result without preceding tool_use").
|
|
candidate = fixToolPairs(candidate);
|
|
candidate = stripTrailingAssistantOrphanToolUse(candidate);
|
|
const tokens = estimateTokens(JSON.stringify(candidate));
|
|
if (tokens <= targetTokens) break;
|
|
keep = Math.max(2, Math.floor(keep * 0.7)); // Drop 30% each iteration
|
|
}
|
|
|
|
let result = [...system, ...nonSystem.slice(-keep)];
|
|
result = fixToolPairs(result);
|
|
result = fixToolAdjacency(result);
|
|
// Re-run pair fix to drop any tool_result whose matching tool_use was removed by
|
|
// fixToolAdjacency (discussion #2410 — orphan tool_result -> upstream 400).
|
|
result = fixToolPairs(result);
|
|
result = stripTrailingAssistantOrphanToolUse(result);
|
|
|
|
// Add summary of dropped messages
|
|
if (keep < nonSystem.length) {
|
|
const dropped = nonSystem.length - keep;
|
|
result.splice(system.length, 0, {
|
|
role: "system",
|
|
content: `[Context compressed: ${dropped} earlier messages removed to fit context window]`,
|
|
});
|
|
}
|
|
|
|
return result;
|
|
}
|
|
|
|
/**
|
|
* Remove orphaned tool_result messages whose preceding tool_use was dropped.
|
|
* Also removes orphaned tool_use messages without a corresponding tool_result.
|
|
*
|
|
* When purifyHistory() drops oldest messages, it can split tool_use/tool_result
|
|
* pairs — keeping the tool_result but dropping the tool_use that initiated it.
|
|
* This causes upstream providers to reject the request with errors like:
|
|
* - Claude: "tool_result message must be preceded by a tool_use message"
|
|
* - OpenAI: "Invalid message format"
|
|
* - Gemini: "Function response without function call"
|
|
*/
|
|
export function fixToolPairs(messages: Record<string, unknown>[]) {
|
|
// Pass 1: Collect all tool_result IDs from user/tool messages
|
|
const toolResultIds = new Set();
|
|
for (const msg of messages) {
|
|
if (msg.role === "tool" && msg.tool_call_id) {
|
|
toolResultIds.add(msg.tool_call_id);
|
|
}
|
|
if (msg.role === "user" && Array.isArray(msg.content)) {
|
|
for (const block of msg.content) {
|
|
if (block.type === "tool_result" && block.tool_use_id) {
|
|
toolResultIds.add(block.tool_use_id);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Pass 2: Filter assistant messages to remove tool_use without tool_result
|
|
// (Exception: keep tool_use if the assistant message is the last message)
|
|
const isLastMessage = (idx: number) => idx === messages.length - 1;
|
|
const filteredMessages = messages.map((msg, idx) => {
|
|
if (msg.role === "assistant" && !isLastMessage(idx)) {
|
|
let modified = false;
|
|
const newMsg = { ...msg };
|
|
|
|
if (Array.isArray(newMsg.tool_calls)) {
|
|
const filteredToolCalls = newMsg.tool_calls.filter(
|
|
(tc: Record<string, unknown>) => !tc.id || toolResultIds.has(tc.id)
|
|
);
|
|
if (filteredToolCalls.length !== newMsg.tool_calls.length) {
|
|
newMsg.tool_calls = filteredToolCalls;
|
|
modified = true;
|
|
}
|
|
}
|
|
|
|
if (Array.isArray(newMsg.content)) {
|
|
const filteredContent = newMsg.content.filter(
|
|
(block: Record<string, unknown>) =>
|
|
block.type !== "tool_use" || !block.id || toolResultIds.has(block.id)
|
|
);
|
|
if (filteredContent.length !== newMsg.content.length) {
|
|
newMsg.content = filteredContent;
|
|
modified = true;
|
|
}
|
|
}
|
|
|
|
return modified ? newMsg : msg;
|
|
}
|
|
return msg;
|
|
});
|
|
|
|
// Pass 3: Collect all remaining tool_use IDs from assistant messages
|
|
const toolCallIds = new Set();
|
|
for (const msg of filteredMessages) {
|
|
if (msg.role === "assistant") {
|
|
if (Array.isArray(msg.tool_calls)) {
|
|
for (const tc of msg.tool_calls) {
|
|
if (tc.id) toolCallIds.add(tc.id);
|
|
}
|
|
}
|
|
if (Array.isArray(msg.content)) {
|
|
for (const block of msg.content) {
|
|
if (block.type === "tool_use" && block.id) {
|
|
toolCallIds.add(block.id);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Pass 4: Filter user/tool messages to remove tool_result without tool_use
|
|
return filteredMessages
|
|
.map((msg) => {
|
|
if (msg.role === "tool" && msg.tool_call_id) {
|
|
if (!toolCallIds.has(msg.tool_call_id)) return null;
|
|
}
|
|
|
|
if (msg.role === "user" && Array.isArray(msg.content)) {
|
|
const filteredContent = msg.content.filter(
|
|
(block: Record<string, unknown>) =>
|
|
block.type !== "tool_result" || !block.tool_use_id || toolCallIds.has(block.tool_use_id)
|
|
);
|
|
if (filteredContent.length !== msg.content.length) {
|
|
if (filteredContent.length === 0) return null;
|
|
return { ...msg, content: filteredContent };
|
|
}
|
|
}
|
|
|
|
// Drop assistant messages if their content AND tool_calls became empty
|
|
if (msg.role === "assistant") {
|
|
const hasContent =
|
|
typeof msg.content === "string"
|
|
? msg.content.trim().length > 0
|
|
: Array.isArray(msg.content) && msg.content.length > 0;
|
|
const hasToolCalls = Array.isArray(msg.tool_calls) && msg.tool_calls.length > 0;
|
|
if (!hasContent && !hasToolCalls) {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
return msg;
|
|
})
|
|
.filter(Boolean) as Record<string, unknown>[];
|
|
}
|
|
|
|
/**
|
|
* Adjacency guard: Claude requires `tool_result` in the IMMEDIATELY NEXT
|
|
* message after `tool_use`, not just somewhere later in the array.
|
|
*
|
|
* `fixToolPairs` checks global ID presence but not adjacency. This function
|
|
* runs after `fixToolPairs` and removes `tool_use` blocks from assistant
|
|
* messages where the next message does not contain a matching `tool_result`.
|
|
*/
|
|
export function fixToolAdjacency(messages: Record<string, unknown>[]): Record<string, unknown>[] {
|
|
if (messages.length <= 1) return messages;
|
|
|
|
const result: Record<string, unknown>[] = [];
|
|
|
|
for (let i = 0; i < messages.length; i++) {
|
|
const msg = messages[i];
|
|
const nextMsg = messages[i + 1];
|
|
|
|
if (msg.role !== "assistant" || !nextMsg) {
|
|
result.push(msg);
|
|
continue;
|
|
}
|
|
|
|
// Collect tool_result IDs from the NEXT message only
|
|
const nextToolResultIds = new Set<string>();
|
|
if (nextMsg.role === "tool" && nextMsg.tool_call_id) {
|
|
nextToolResultIds.add(String(nextMsg.tool_call_id));
|
|
}
|
|
if (nextMsg.role === "user" && Array.isArray(nextMsg.content)) {
|
|
for (const block of nextMsg.content as Record<string, unknown>[]) {
|
|
if (block.type === "tool_result" && block.tool_use_id) {
|
|
nextToolResultIds.add(String(block.tool_use_id));
|
|
}
|
|
}
|
|
}
|
|
|
|
let modified = false;
|
|
const newMsg: Record<string, unknown> = { ...msg };
|
|
|
|
// Filter tool_use blocks in content array (Claude format)
|
|
if (Array.isArray(newMsg.content)) {
|
|
const filteredContent = (newMsg.content as Record<string, unknown>[]).filter(
|
|
(block) => block.type !== "tool_use" || !block.id || nextToolResultIds.has(String(block.id))
|
|
);
|
|
if (filteredContent.length !== (newMsg.content as unknown[]).length) {
|
|
newMsg.content = filteredContent;
|
|
modified = true;
|
|
}
|
|
}
|
|
|
|
// Filter tool_calls array (OpenAI format) — independently of content
|
|
if (Array.isArray(newMsg.tool_calls)) {
|
|
const filteredToolCalls = (newMsg.tool_calls as Record<string, unknown>[]).filter(
|
|
(tc: Record<string, unknown>) => !tc.id || nextToolResultIds.has(String(tc.id))
|
|
);
|
|
if (filteredToolCalls.length !== (newMsg.tool_calls as unknown[]).length) {
|
|
newMsg.tool_calls = filteredToolCalls;
|
|
modified = true;
|
|
}
|
|
}
|
|
|
|
if (modified) {
|
|
// Drop assistant message if it became empty
|
|
const hasContent =
|
|
typeof newMsg.content === "string"
|
|
? (newMsg.content as string).trim().length > 0
|
|
: Array.isArray(newMsg.content) && (newMsg.content as unknown[]).length > 0;
|
|
const hasToolCalls = Array.isArray(newMsg.tool_calls) && newMsg.tool_calls.length > 0;
|
|
if (!hasContent && !hasToolCalls) continue;
|
|
result.push(newMsg);
|
|
} else {
|
|
result.push(msg);
|
|
}
|
|
}
|
|
|
|
return result;
|
|
}
|
|
|
|
/**
|
|
* Upstream-send guard: after `fixToolPairs`, strip a trailing assistant
|
|
* message whose only/remaining content is an orphan `tool_use` block.
|
|
*
|
|
* `fixToolPairs` intentionally preserves a final-message `tool_use` because
|
|
* during context pruning the client is still waiting on the matching
|
|
* `tool_result` — dropping it there would lose state. But on the
|
|
* upstream-send path the request body must end on a user turn; a trailing
|
|
* `assistant(tool_use)` triggers the same Anthropic 400 the guard is
|
|
* trying to prevent:
|
|
* messages.N: `tool_use` ids were found without `tool_result` blocks
|
|
* immediately after: toolu_...
|
|
*
|
|
* Behavior:
|
|
* - If the last message is `assistant` and contains any `tool_use` block,
|
|
* those blocks are removed.
|
|
* - If removal leaves the message with no content / tool_calls at all, the
|
|
* message itself is dropped.
|
|
* - Idempotent on clean histories (trailing user, trailing assistant with
|
|
* only text/thinking, etc.).
|
|
*/
|
|
export function stripTrailingAssistantOrphanToolUse(
|
|
messages: Record<string, unknown>[]
|
|
): Record<string, unknown>[] {
|
|
if (!Array.isArray(messages) || messages.length === 0) return messages;
|
|
|
|
const lastIdx = messages.length - 1;
|
|
const last = messages[lastIdx];
|
|
if (!last || last.role !== "assistant") return messages;
|
|
|
|
let modified = false;
|
|
const newLast: Record<string, unknown> = { ...last };
|
|
|
|
if (Array.isArray(newLast.tool_calls)) {
|
|
const filteredCalls = (newLast.tool_calls as Record<string, unknown>[]).filter(
|
|
() => false // remove all trailing tool_calls (none can be paired by definition)
|
|
);
|
|
if (filteredCalls.length !== (newLast.tool_calls as unknown[]).length) {
|
|
newLast.tool_calls = filteredCalls;
|
|
modified = true;
|
|
}
|
|
}
|
|
|
|
if (Array.isArray(newLast.content)) {
|
|
const filteredContent = (newLast.content as Record<string, unknown>[]).filter(
|
|
(block) => block.type !== "tool_use"
|
|
);
|
|
if (filteredContent.length !== (newLast.content as unknown[]).length) {
|
|
newLast.content = filteredContent;
|
|
modified = true;
|
|
}
|
|
}
|
|
|
|
if (!modified) return messages;
|
|
|
|
// If the last message is now empty, drop it.
|
|
const hasContent =
|
|
typeof newLast.content === "string"
|
|
? (newLast.content as string).trim().length > 0
|
|
: Array.isArray(newLast.content) && (newLast.content as unknown[]).length > 0;
|
|
const hasToolCalls =
|
|
Array.isArray(newLast.tool_calls) && (newLast.tool_calls as unknown[]).length > 0;
|
|
|
|
const result = messages.slice(0, lastIdx);
|
|
if (hasContent || hasToolCalls) result.push(newLast);
|
|
return result;
|
|
}
|