From cad308942b70a6b75fdbe41e89e7fde02724fc44 Mon Sep 17 00:00:00 2001 From: adevwithpurpose Date: Sat, 15 Aug 2026 19:12:36 -0300 Subject: [PATCH] fix(combo): surface context-overflow before compression so oversized requests fail fast with a clear error (#10225) --- ...mbo-context-overflow-before-compression.md | 1 + open-sse/services/combo.ts | 14 ++ open-sse/services/combo/dispatchPrelude.ts | 10 + .../services/combo/knownContextOverflow.ts | 41 ++++- open-sse/services/combo/targetResolution.ts | 6 + open-sse/services/combo/types.ts | 10 + src/sse/handlers/chat.ts | 35 ++++ ...context-overflow-compression-probe.test.ts | 173 ++++++++++++++++++ 8 files changed, 289 insertions(+), 1 deletion(-) create mode 100644 changelog.d/fixes/10225-combo-context-overflow-before-compression.md create mode 100644 tests/unit/combo-context-overflow-compression-probe.test.ts diff --git a/changelog.d/fixes/10225-combo-context-overflow-before-compression.md b/changelog.d/fixes/10225-combo-context-overflow-before-compression.md new file mode 100644 index 0000000000..0a678180af --- /dev/null +++ b/changelog.d/fixes/10225-combo-context-overflow-before-compression.md @@ -0,0 +1 @@ +- **fix(combo):** defer the known-context-overflow hard rejection for compressible requests so compression runs before the final context gate, instead of a raw-body estimate 400'ing generic Responses clients targeting a large model before OmniRoute can shrink it ([#10225](https://github.com/diegosouzapw/OmniRoute/issues/10225)) \ No newline at end of file diff --git a/open-sse/services/combo.ts b/open-sse/services/combo.ts index df86a79375..c6f6045285 100644 --- a/open-sse/services/combo.ts +++ b/open-sse/services/combo.ts @@ -591,6 +591,8 @@ export async function handleComboChat({ nesting = null, hiddenModelsByProvider = getHiddenModelsByProvider(), clientManagedResponsesContext = false, + deferContextOverflowWhenCompressible = false, + compressionExclusions, }: HandleComboChatOptions): Promise { const comboCtx = createComboContext({ body, combo, settings, relayOptions, log }); const { @@ -651,6 +653,8 @@ export async function handleComboChat({ signal, apiKeyAllowedConnections, hiddenModelsByProvider, + deferContextOverflowWhenCompressible, + compressionExclusions, runCombo: handleComboChat, }); if (fusionDispatch) return fusionDispatch; @@ -700,6 +704,8 @@ export async function handleComboChat({ signal, apiKeyAllowedConnections, hiddenModelsByProvider, + deferContextOverflowWhenCompressible, + compressionExclusions, runCombo: handleComboChat, }); if (runtimeUnitDispatch) return runtimeUnitDispatch; @@ -723,6 +729,8 @@ export async function handleComboChat({ signal, hiddenModelsByProvider, clientManagedResponsesContext, + deferContextOverflowWhenCompressible, + compressionExclusions, relayOptions, }); } @@ -750,6 +758,8 @@ export async function handleComboChat({ buildAutoCandidates, hiddenModelsByProvider, clientManagedResponsesContext, + deferContextOverflowWhenCompressible, + compressionExclusions, }); if ("earlyResponse" in targetResolution) return targetResolution.earlyResponse; const { stickyWeightedLimit, getWeightedStepKeyForTarget, preScreenMap } = targetResolution; @@ -2441,6 +2451,8 @@ async function handleRoundRobinCombo({ nesting = null, hiddenModelsByProvider = getHiddenModelsByProvider(), clientManagedResponsesContext, + deferContextOverflowWhenCompressible = false, + compressionExclusions, relayOptions, }: HandleRoundRobinOptions): Promise { const config = settings @@ -2498,6 +2510,8 @@ async function handleRoundRobinCombo({ const evalRankedTargets = orderTargetsByEvalScores(tagFilteredTargets, config.evalRouting, log); const knownContextOverflow = getKnownContextOverflow(evalRankedTargets, body, { clientManagedResponsesContext, + deferContextOverflowWhenCompressible, + compressionExclusions, }); if (knownContextOverflow) { return errorResponseWithComboDiagnostics( diff --git a/open-sse/services/combo/dispatchPrelude.ts b/open-sse/services/combo/dispatchPrelude.ts index 7caf82fe76..c205dae8f9 100644 --- a/open-sse/services/combo/dispatchPrelude.ts +++ b/open-sse/services/combo/dispatchPrelude.ts @@ -76,6 +76,10 @@ type PreludeBaseOptionArgs = { apiKeyAllowedConnections?: string[] | null; hiddenModelsByProvider?: HiddenModelsByProvider; clientManagedResponsesContext?: boolean; + /** #10225 — defer the hard context-overflow preflight when compression is enabled. */ + deferContextOverflowWhenCompressible?: boolean; + /** Server-side compression exclusions (#8034). */ + compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions; }; /** Rebuild handleComboChat's option bag verbatim for a recursive dispatch. */ @@ -93,6 +97,8 @@ function buildBaseOptions(a: PreludeBaseOptionArgs): HandleComboChatOptions { apiKeyAllowedConnections: a.apiKeyAllowedConnections, hiddenModelsByProvider: a.hiddenModelsByProvider, clientManagedResponsesContext: a.clientManagedResponsesContext, + deferContextOverflowWhenCompressible: a.deferContextOverflowWhenCompressible, + compressionExclusions: a.compressionExclusions, }; } @@ -366,6 +372,8 @@ export async function tryFusionDispatch(args: { signal?: AbortSignal | null; apiKeyAllowedConnections?: string[] | null; hiddenModelsByProvider?: HiddenModelsByProvider; + deferContextOverflowWhenCompressible?: boolean; + compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions; runCombo: RunCombo; }): Promise { const { cfg, combo, config, strategy, log } = args; @@ -589,6 +597,8 @@ export async function tryRuntimeUnitDispatch(args: { signal?: AbortSignal | null; apiKeyAllowedConnections?: string[] | null; hiddenModelsByProvider?: HiddenModelsByProvider; + deferContextOverflowWhenCompressible?: boolean; + compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions; runCombo: RunCombo; }): Promise { const { body, combo, config, strategy, allCombos, log, settings } = args; diff --git a/open-sse/services/combo/knownContextOverflow.ts b/open-sse/services/combo/knownContextOverflow.ts index db9cb2d602..645266e71f 100644 --- a/open-sse/services/combo/knownContextOverflow.ts +++ b/open-sse/services/combo/knownContextOverflow.ts @@ -17,6 +17,7 @@ */ import { getResolvedModelCapabilities } from "../modelCapabilities.ts"; +import { isCompressionExcluded, type CompressionExclusions } from "../compression/exclusions.ts"; import { deriveRequestCompatibilityRequirements } from "./comboStructure.ts"; import type { ResolvedComboTarget } from "./types.ts"; @@ -28,6 +29,19 @@ export type KnownContextOverflow = { targetCount: number; }; +export type KnownContextOverflowOptions = { + clientManagedResponsesContext?: boolean; + /** + * When prompt compression is enabled for this request (global compression switch + * AND not API-key opted-out), defer the hard preflight so chatCore's compression + * pipeline runs before the final context gate — instead of a raw-body estimate + * rejecting a compressible request up front. (#10225) + */ + deferContextOverflowWhenCompressible?: boolean; + /** Server-side compression exclusions (#8034) — targets matching one cannot run compression. */ + compressionExclusions?: CompressionExclusions; +}; + // #7177: an empty array/object (e.g. a default `messages: []` some combo entrypoints inject // when the caller sent none) has no real content — counting it would charge a few phantom // "structural" tokens (JSON.stringify braces/brackets) toward the estimate, which is enough @@ -69,7 +83,7 @@ export function getKnownContextLimit( export function getKnownContextOverflow( targets: ResolvedComboTarget[], body: Record, - options: { clientManagedResponsesContext?: boolean } = {} + options: KnownContextOverflowOptions = {} ): KnownContextOverflow | null { if (targets.length === 0) return null; // Native Codex Responses clients compact their own item history. Let the concrete @@ -85,6 +99,31 @@ export function getKnownContextOverflow( ) { return null; } + // #10225: a conservative raw-body context estimate must not be treated as proof + // that a compression-enabled request cannot fit. When compression is available + // for this request AND at least one target can actually run it, defer the hard + // rejection so handleChatCore runs proactive compression (chatCore.ts) and its + // post-compression enforceOutputTokenBudget becomes the final context gate — + // returning a local `context_length_exceeded` only if the compressed body still + // cannot fit (no upstream dispatch). Each excluded/native-codex-passthrough + // target is skipped; if no target can compress, the fast preflight is kept. + if ( + options.deferContextOverflowWhenCompressible === true && + targets.some( + (target) => + !isCompressionExcluded( + { + provider: target.provider, + model: target.modelStr.includes("/") + ? target.modelStr.split("/").slice(1).join("/") + : target.modelStr, + }, + options.compressionExclusions + ) + ) + ) { + return null; + } const requirements = deriveRequestCompatibilityRequirements(body); if (requirements.requiredContextTokens <= 0) return null; diff --git a/open-sse/services/combo/targetResolution.ts b/open-sse/services/combo/targetResolution.ts index d5af5a6e01..0ef355b15a 100644 --- a/open-sse/services/combo/targetResolution.ts +++ b/open-sse/services/combo/targetResolution.ts @@ -115,6 +115,10 @@ export interface ResolveComboTargetPipelineDeps { hiddenModelsByProvider?: HiddenModelsByProvider; /** Native Responses clients (for example Codex CLI/Desktop) manage compaction themselves. */ clientManagedResponsesContext?: boolean; + /** #10225 — defer the hard context-overflow preflight when compression is enabled for this request. */ + deferContextOverflowWhenCompressible?: boolean; + /** Server-side compression exclusions (#8034) — which targets can run compression. */ + compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions; } export interface ResolvedComboTargetPipeline { @@ -730,6 +734,8 @@ export async function resolveComboTargetPipeline( const overflow = getKnownContextOverflow(orderedTargets, body, { clientManagedResponsesContext: deps.clientManagedResponsesContext, + deferContextOverflowWhenCompressible: deps.deferContextOverflowWhenCompressible, + compressionExclusions: deps.compressionExclusions, }); if (overflow) { return { earlyResponse: buildContextOverflowResponse(overflow, orderedTargets, log) }; diff --git a/open-sse/services/combo/types.ts b/open-sse/services/combo/types.ts index 9f9f31c4b4..aa5cd0772a 100644 --- a/open-sse/services/combo/types.ts +++ b/open-sse/services/combo/types.ts @@ -6,6 +6,7 @@ * — logic unchanged, re-exported from combo.ts for backward compatibility. */ +import type { CompressionExclusions } from "../compression/exclusions.ts"; import type { ProviderCandidate } from "../autoCombo/scoring.ts"; export const RESET_WINDOW_NAMES = ["weekly", "session", "monthly"] as const; @@ -112,6 +113,15 @@ export type HandleComboChatOptions = { hiddenModelsByProvider?: HiddenModelsByProvider; /** Native Responses clients (for example Codex CLI/Desktop) manage compaction themselves. */ clientManagedResponsesContext?: boolean; + /** + * #10225: request-scoped flag — prompt compression is enabled for this request + * (global compression switch ON and not opted-out by the API key). When set, the + * combo preflight defers its hard context-overflow rejection so chatCore's + * compression runs before the final context gate. + */ + deferContextOverflowWhenCompressible?: boolean; + /** Server-side compression exclusions (#8034) — used to check which targets can run compression. */ + compressionExclusions?: CompressionExclusions; }; export type HandleRoundRobinOptions = Omit; diff --git a/src/sse/handlers/chat.ts b/src/sse/handlers/chat.ts index 9002225a1b..458571ab9d 100644 --- a/src/sse/handlers/chat.ts +++ b/src/sse/handlers/chat.ts @@ -33,6 +33,8 @@ import type { SingleModelTarget } from "@omniroute/open-sse/services/combo/types import { mergeAbortSignals } from "@omniroute/open-sse/executors/base.ts"; import { resolveRequestAutoControls } from "@omniroute/open-sse/services/autoCombo/requestControls.ts"; import { isVerifiedNativeCodexRequest } from "@omniroute/open-sse/config/codexIdentity.ts"; +import { resolveCompressionSettings } from "@omniroute/open-sse/handlers/chatCore/compressionSettings.ts"; +import type { CompressionExclusions } from "@omniroute/open-sse/services/compression/exclusions.ts"; import { resolveComboConfig } from "@omniroute/open-sse/services/comboConfig.ts"; import { injectHandoffIntoBody } from "@omniroute/open-sse/services/contextHandoff.ts"; import { @@ -209,6 +211,31 @@ let combosCacheTs = 0; let combosCacheVersionSnapshot = -1; const COMBOS_CACHE_TTL_MS = 10_000; +/** + * #10225 — resolve whether this request's combo preflight should DEFER its hard + * context-overflow rejection so chatCore's compression runs first. + * + * Mirrors handleChatCore's own enablement determination (chatCore.ts): defer only + * when the global compression switch is ON and the API key has not opted out + * (`apiKeyInfo.compressionEnabled !== false`). Per-target applicability (server-side + * exclusions) is checked inside getKnownContextOverflow via the returned exclusions. + * Fail closed (defer=false) on any lookup error — the existing hard preflight stays. + */ +async function resolveComboContextOverflowDeferral( + logger: { warn?: (...args: unknown[]) => void } | null | undefined, + apiKeyInfo: { compressionEnabled?: boolean } | null | undefined +): Promise<{ defer: boolean; exclusions: CompressionExclusions | undefined }> { + try { + const compression = await resolveCompressionSettings(logger); + return { + defer: compression.enabled && apiKeyInfo?.compressionEnabled !== false, + exclusions: compression.settings?.exclusions, + }; + } catch { + return { defer: false, exclusions: undefined }; + } +} + async function getCombosCachedForChat(): Promise { const now = Date.now(); // Explicit non-null check: we intentionally cache and return the Promise @@ -824,9 +851,13 @@ async function handleChatImplementation( // Context-relay keeps generation in combo.ts, but handoff injection lives here // because only this layer knows which connectionId was actually selected. + const { defer: deferContextOverflowWhenCompressible, exclusions: compressionExclusions } = + await resolveComboContextOverflowDeferral(log, apiKeyInfo); const response = await (handleComboChat as any)({ body, combo, + deferContextOverflowWhenCompressible, + compressionExclusions, clientManagedResponsesContext: sourceFormat === "openai-responses" && new URL(request.url).pathname.split("/").includes("responses") && @@ -1103,9 +1134,13 @@ async function handleSingleModelChat( ); log.info("ROUTING", `Auto-combo redirect from handleSingleModelChat for "${modelStr}"`); log.info("ROUTING", `Auto-combo redirect to combo flow for "${modelStr}"`); + const { defer: sNetDefer, exclusions: sNetExclusions } = + await resolveComboContextOverflowDeferral(log, apiKeyInfo); return handleComboChat({ body, combo: redirectCombo, + deferContextOverflowWhenCompressible: sNetDefer, + compressionExclusions: sNetExclusions, clientManagedResponsesContext: detectFormatFromEndpoint(body, clientRawRequest?.endpoint || "") === "openai-responses" && String(clientRawRequest?.endpoint || "") diff --git a/tests/unit/combo-context-overflow-compression-probe.test.ts b/tests/unit/combo-context-overflow-compression-probe.test.ts new file mode 100644 index 0000000000..7ab8604735 --- /dev/null +++ b/tests/unit/combo-context-overflow-compression-probe.test.ts @@ -0,0 +1,173 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +/** + * #10225 — combo known-context-overflow must NOT hard-reject a compressible + * request before OmniRoute's compression pipeline can run. + * + * Root cause: getKnownContextOverflow() estimates the RAW body (ceil(serializedChars/4) + * over the whole Responses input[]) during combo target resolution, before any + * compression. When every known target limit is below that raw estimate, both call + * sites (round-robin + target-resolution) convert it into an immediate local 400 + * `context_length_exceeded` with attempted:0 — so chatCore's proactive compression + * (which can shrink 294133→111529, 62% in the reporter's case) never runs. The only + * existing bypass (clientManagedResponsesContext) is gated to VERIFIED native Codex + * clients, so a generic Responses client (e.g. OpenCode) pointed at a codex model + * still hits the hard gate. + * + * Fix: thread a request-scoped `deferContextOverflowWhenCompressible` flag (set when + * the global compression switch is ON and not API-key opted-out). When set AND at + * least one target can run compression, getKnownContextOverflow returns null so the + * request reaches chatCore, whose post-compression enforceOutputTokenBudget becomes + * the final context gate — a local 400 only if the compressed body still cannot fit. + */ + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-combo-overflow-compress-")); +const ORIGINAL_DATA_DIR = process.env.DATA_DIR; +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const { saveModelsDevCapabilities, clearModelsDevCapabilities } = + await import("../../src/lib/modelsDevSync.ts"); +const { getKnownContextOverflow, handleComboChat } = await import( + "../../open-sse/services/combo.ts" +); + +test.after(() => { + core.resetDbInstance(); + if (ORIGINAL_DATA_DIR === undefined) { + delete process.env.DATA_DIR; + } else { + process.env.DATA_DIR = ORIGINAL_DATA_DIR; + } + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test.beforeEach(() => { + clearModelsDevCapabilities(); +}); + +function capabilityEntry(limitContext: number | null) { + return { + tool_call: true, + reasoning: false, + attachment: false, + structured_output: true, + temperature: true, + modalities_input: JSON.stringify(["text"]), + modalities_output: JSON.stringify(["text"]), + knowledge_cutoff: null, + release_date: null, + last_updated: null, + status: null, + family: null, + open_weights: false, + limit_context: limitContext, + limit_input: limitContext, + limit_output: 4096, + interleaved_field: null, + }; +} + +function target(modelStr: string) { + return { + kind: "model" as const, + stepId: modelStr, + executionKey: modelStr, + modelStr, + provider: modelStr.includes("/") ? modelStr.split("/")[0] : modelStr, + providerId: null, + connectionId: null, + weight: 1, + label: null, + }; +} + +// A generic Responses-API body whose estimate lands near `tokens` tokens (4 chars/token). +// Uses `input:` (not `messages:`) to mirror the OpenCode/Codex Responses surface. +function bigResponsesBody(tokens: number) { + return { input: [["user", "x".repeat(tokens * 4)]] }; +} + +const noopLog = { info() {}, warn() {}, error() {}, debug() {} }; + +test("#10225 getKnownContextOverflow defers the hard overflow when compression is available", () => { + saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } }); + const body = bigResponsesBody(275_000); + + // Compression enabled + target can compress -> defer (null). + assert.equal( + getKnownContextOverflow([target("codex/gpt-5.6-terra")], body, { + deferContextOverflowWhenCompressible: true, + }), + null, + "compressible request must defer so chatCore compression can run (#10225)" + ); + + // Compression disabled -> the existing hard overflow is preserved (never lose #7177). + const hard = getKnownContextOverflow([target("codex/gpt-5.6-terra")], body); + assert.ok(hard); + assert.ok(hard.requiredContextTokens > hard.maxKnownContextTokens); + + // Compression enabled but EVERY target is excluded from compression -> keep the hard gate. + const excluded = getKnownContextOverflow([target("codex/gpt-5.6-terra")], body, { + deferContextOverflowWhenCompressible: true, + compressionExclusions: ["gpt-5.6-terra"], + }); + assert.ok(excluded, "fully-excluded targets must retain the hard preflight"); +}); + +test("#10225 combo does not early-400 a compressible over-limit request when deferral is on", async () => { + saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } }); + let dispatches = 0; + + const response = await handleComboChat({ + body: bigResponsesBody(275_000), + combo: { + name: "codex-compress-overflow", + strategy: "priority", + models: ["codex/gpt-5.6-terra"], + }, + deferContextOverflowWhenCompressible: true, + clientManagedResponsesContext: false, + isModelAvailable: async () => true, + handleSingleModel: async () => { + dispatches += 1; + return new Response("ok", { status: 200 }); + }, + log: noopLog, + }); + + assert.notEqual(response.status, 400, "compression-enabled request must reach chatCore"); + assert.equal(dispatches, 1, "must dispatch so chatCore compaction runs first"); +}); + +test("#10225 combo keeps the fast 400 when compression is disabled", async () => { + saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } }); + let dispatches = 0; + + const response = await handleComboChat({ + body: bigResponsesBody(275_000), + combo: { + name: "codex-compress-disabled", + strategy: "priority", + models: ["codex/gpt-5.6-terra"], + }, + deferContextOverflowWhenCompressible: false, + clientManagedResponsesContext: false, + isModelAvailable: async () => true, + handleSingleModel: async () => { + dispatches += 1; + return new Response("ok", { status: 200 }); + }, + log: noopLog, + }); + + assert.equal(response.status, 400); + assert.equal(dispatches, 0, "#7177 anti-exhaustion guard must survive when compression is off"); + const body = await response.json(); + assert.equal(body.error.code, "context_length_exceeded"); +});