Compare commits

...

2 Commits

Author SHA1 Message Date
adevwithpurpose
8d473172c1 fix(combo): make context-overflow deferral target-aware for native Codex passthrough (#10225)
The deferral added by the prior commit checked only operator-named
compression exclusions when deciding whether at least one target "can
compress" — it never accounted for native Codex Responses passthrough
targets, which chatCore.ts unconditionally excludes from compression
(compressionExcluded = nativeCodexPassthrough || ...). Deferring on such
a target's account let an oversized request skip both the combo preflight
AND compression, reaching fetch() uncompressed.

Thread the same request-shape facts chatCore.ts uses
(shouldUseNativeCodexPassthrough: provider/sourceFormat/endpointPath/body/
headers) down into getKnownContextOverflow so the deferral decision can
never drift from chatCore's own — a native-codex-passthrough target now
never counts as "compressible", so a pool made only of such targets keeps
the fast local 400 instead of a wasted round trip.

Adds regression coverage: the pure getKnownContextOverflow target-aware
check, an end-to-end handleComboChat proof that a native-codex-only pool
fails fast with zero dispatches, and two real handleChatCore-path tests
proving compression actually reduces the dispatched body when eligible,
and that a still-too-large-after-compression request is rejected locally
without an upstream call.
2026-08-17 23:10:01 -03:00
adevwithpurpose
cad308942b fix(combo): surface context-overflow before compression so oversized requests fail fast with a clear error (#10225) 2026-08-15 19:12:36 -03:00
8 changed files with 626 additions and 2 deletions

View File

@@ -0,0 +1 @@
- **fix(combo):** defer the known-context-overflow hard rejection for compressible requests so compression runs before the final context gate, instead of a raw-body estimate 400'ing generic Responses clients targeting a large model before OmniRoute can shrink it ([#10225](https://github.com/diegosouzapw/OmniRoute/issues/10225))

View File

@@ -591,6 +591,11 @@ export async function handleComboChat({
nesting = null,
hiddenModelsByProvider = getHiddenModelsByProvider(),
clientManagedResponsesContext = false,
deferContextOverflowWhenCompressible = false,
compressionExclusions,
sourceFormat = null,
endpointPath = null,
requestHeaders = null,
}: HandleComboChatOptions): Promise<Response> {
const comboCtx = createComboContext({ body, combo, settings, relayOptions, log });
const {
@@ -651,6 +656,11 @@ export async function handleComboChat({
signal,
apiKeyAllowedConnections,
hiddenModelsByProvider,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
runCombo: handleComboChat,
});
if (fusionDispatch) return fusionDispatch;
@@ -700,6 +710,11 @@ export async function handleComboChat({
signal,
apiKeyAllowedConnections,
hiddenModelsByProvider,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
runCombo: handleComboChat,
});
if (runtimeUnitDispatch) return runtimeUnitDispatch;
@@ -723,6 +738,11 @@ export async function handleComboChat({
signal,
hiddenModelsByProvider,
clientManagedResponsesContext,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
relayOptions,
});
}
@@ -750,6 +770,11 @@ export async function handleComboChat({
buildAutoCandidates,
hiddenModelsByProvider,
clientManagedResponsesContext,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
});
if ("earlyResponse" in targetResolution) return targetResolution.earlyResponse;
const { stickyWeightedLimit, getWeightedStepKeyForTarget, preScreenMap } = targetResolution;
@@ -2441,6 +2466,11 @@ async function handleRoundRobinCombo({
nesting = null,
hiddenModelsByProvider = getHiddenModelsByProvider(),
clientManagedResponsesContext,
deferContextOverflowWhenCompressible = false,
compressionExclusions,
sourceFormat = null,
endpointPath = null,
requestHeaders = null,
relayOptions,
}: HandleRoundRobinOptions): Promise<Response> {
const config = settings
@@ -2498,6 +2528,11 @@ async function handleRoundRobinCombo({
const evalRankedTargets = orderTargetsByEvalScores(tagFilteredTargets, config.evalRouting, log);
const knownContextOverflow = getKnownContextOverflow(evalRankedTargets, body, {
clientManagedResponsesContext,
deferContextOverflowWhenCompressible,
compressionExclusions,
sourceFormat,
endpointPath,
requestHeaders,
});
if (knownContextOverflow) {
return errorResponseWithComboDiagnostics(

View File

@@ -76,6 +76,14 @@ type PreludeBaseOptionArgs = {
apiKeyAllowedConnections?: string[] | null;
hiddenModelsByProvider?: HiddenModelsByProvider;
clientManagedResponsesContext?: boolean;
/** #10225 — defer the hard context-overflow preflight when compression is enabled. */
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034). */
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
/** #10503 — request-shape facts for the target-aware deferral check (see knownContextOverflow.ts). */
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
};
/** Rebuild handleComboChat's option bag verbatim for a recursive dispatch. */
@@ -93,6 +101,11 @@ function buildBaseOptions(a: PreludeBaseOptionArgs): HandleComboChatOptions {
apiKeyAllowedConnections: a.apiKeyAllowedConnections,
hiddenModelsByProvider: a.hiddenModelsByProvider,
clientManagedResponsesContext: a.clientManagedResponsesContext,
deferContextOverflowWhenCompressible: a.deferContextOverflowWhenCompressible,
compressionExclusions: a.compressionExclusions,
sourceFormat: a.sourceFormat,
endpointPath: a.endpointPath,
requestHeaders: a.requestHeaders,
};
}
@@ -366,6 +379,11 @@ export async function tryFusionDispatch(args: {
signal?: AbortSignal | null;
apiKeyAllowedConnections?: string[] | null;
hiddenModelsByProvider?: HiddenModelsByProvider;
deferContextOverflowWhenCompressible?: boolean;
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
runCombo: RunCombo;
}): Promise<Response | null> {
const { cfg, combo, config, strategy, log } = args;
@@ -589,6 +607,11 @@ export async function tryRuntimeUnitDispatch(args: {
signal?: AbortSignal | null;
apiKeyAllowedConnections?: string[] | null;
hiddenModelsByProvider?: HiddenModelsByProvider;
deferContextOverflowWhenCompressible?: boolean;
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
runCombo: RunCombo;
}): Promise<Response | null> {
const { body, combo, config, strategy, allCombos, log, settings } = args;

View File

@@ -17,6 +17,8 @@
*/
import { getResolvedModelCapabilities } from "../modelCapabilities.ts";
import { isCompressionExcluded, type CompressionExclusions } from "../compression/exclusions.ts";
import { shouldUseNativeCodexPassthrough } from "../../handlers/chatCore/passthroughHelpers.ts";
import { deriveRequestCompatibilityRequirements } from "./comboStructure.ts";
import type { ResolvedComboTarget } from "./types.ts";
@@ -28,6 +30,29 @@ export type KnownContextOverflow = {
targetCount: number;
};
export type KnownContextOverflowOptions = {
clientManagedResponsesContext?: boolean;
/**
* When prompt compression is enabled for this request (global compression switch
* AND not API-key opted-out), defer the hard preflight so chatCore's compression
* pipeline runs before the final context gate — instead of a raw-body estimate
* rejecting a compressible request up front. (#10225)
*/
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034) — targets matching one cannot run compression. */
compressionExclusions?: CompressionExclusions;
/**
* #10503: the exact request-shape facts chatCore.ts uses to decide
* `shouldUseNativeCodexPassthrough` (open-sse/handlers/chatCore/passthroughHelpers.ts) —
* threaded down so the deferral decision below can be target-aware instead of
* relying on the looser `clientManagedResponsesContext` proxy. Reused verbatim
* (not re-derived) so the combo-layer decision can never drift from chatCore's own.
*/
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
};
// #7177: an empty array/object (e.g. a default `messages: []` some combo entrypoints inject
// when the caller sent none) has no real content — counting it would charge a few phantom
// "structural" tokens (JSON.stringify braces/brackets) toward the estimate, which is enough
@@ -69,7 +94,7 @@ export function getKnownContextLimit(
export function getKnownContextOverflow(
targets: ResolvedComboTarget[],
body: Record<string, unknown>,
options: { clientManagedResponsesContext?: boolean } = {}
options: KnownContextOverflowOptions = {}
): KnownContextOverflow | null {
if (targets.length === 0) return null;
// Native Codex Responses clients compact their own item history. Let the concrete
@@ -85,6 +110,55 @@ export function getKnownContextOverflow(
) {
return null;
}
// #10225 / #10499-sweep #10503: a conservative raw-body context estimate must not
// be treated as proof that a compression-enabled request cannot fit. When
// compression is available for this request AND at least one target can actually
// run it, defer the hard rejection so handleChatCore runs proactive compression
// (chatCore.ts) and its post-compression enforceOutputTokenBudget becomes the
// final context gate — returning a local `context_length_exceeded` only if the
// compressed body still cannot fit (no upstream dispatch).
//
// Target-awareness is load-bearing here: a target is only a valid reason to defer
// when handleChatCore will ACTUALLY attempt compression for it. Two classes are
// excluded from "can compress" even though `isCompressionExcluded` (operator
// exclusions) says nothing about them:
// - Operator-excluded targets (#8034, existing `isCompressionExcluded` check).
// - Native Codex Responses passthrough targets: chatCore.ts unconditionally sets
// `compressionExcluded = nativeCodexPassthrough || ...` for these, computed via
// `shouldUseNativeCodexPassthrough()` (chatCore/passthroughHelpers.ts) — called
// here with the SAME request-shape facts (sourceFormat/endpointPath/headers)
// chatCore itself uses, reused verbatim rather than re-derived from the looser
// `clientManagedResponsesContext` flag (which always requires a VERIFIED native
// client; chatCore's own gate does NOT for provider==="codex" — see
// shouldUseNativeCodexPassthrough's `provider === "codex" || isVerifiedNativeCodexRequest`
// short-circuit). Deferring on such a target's account would let an oversized
// body sail straight through to `fetch()` uncompressed instead of being caught
// by either preflight — silently defeating the whole point of this feature.
// If NO target can compress, the fast raw-body preflight is kept (unchanged).
if (
options.deferContextOverflowWhenCompressible === true &&
targets.some((target) => {
const isNativeCodexPassthroughTarget = shouldUseNativeCodexPassthrough({
provider: target.provider,
sourceFormat: options.sourceFormat,
endpointPath: options.endpointPath,
body,
headers: options.requestHeaders,
});
if (isNativeCodexPassthroughTarget) return false;
return !isCompressionExcluded(
{
provider: target.provider,
model: target.modelStr.includes("/")
? target.modelStr.split("/").slice(1).join("/")
: target.modelStr,
},
options.compressionExclusions
);
})
) {
return null;
}
const requirements = deriveRequestCompatibilityRequirements(body);
if (requirements.requiredContextTokens <= 0) return null;

View File

@@ -115,6 +115,14 @@ export interface ResolveComboTargetPipelineDeps {
hiddenModelsByProvider?: HiddenModelsByProvider;
/** Native Responses clients (for example Codex CLI/Desktop) manage compaction themselves. */
clientManagedResponsesContext?: boolean;
/** #10225 — defer the hard context-overflow preflight when compression is enabled for this request. */
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034) — which targets can run compression. */
compressionExclusions?: import("../compression/exclusions.ts").CompressionExclusions;
/** #10503 — request-shape facts for the target-aware deferral check (see knownContextOverflow.ts). */
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
}
export interface ResolvedComboTargetPipeline {
@@ -730,6 +738,11 @@ export async function resolveComboTargetPipeline(
const overflow = getKnownContextOverflow(orderedTargets, body, {
clientManagedResponsesContext: deps.clientManagedResponsesContext,
deferContextOverflowWhenCompressible: deps.deferContextOverflowWhenCompressible,
compressionExclusions: deps.compressionExclusions,
sourceFormat: deps.sourceFormat,
endpointPath: deps.endpointPath,
requestHeaders: deps.requestHeaders,
});
if (overflow) {
return { earlyResponse: buildContextOverflowResponse(overflow, orderedTargets, log) };

View File

@@ -6,6 +6,7 @@
* — logic unchanged, re-exported from combo.ts for backward compatibility.
*/
import type { CompressionExclusions } from "../compression/exclusions.ts";
import type { ProviderCandidate } from "../autoCombo/scoring.ts";
export const RESET_WINDOW_NAMES = ["weekly", "session", "monthly"] as const;
@@ -112,6 +113,25 @@ export type HandleComboChatOptions = {
hiddenModelsByProvider?: HiddenModelsByProvider;
/** Native Responses clients (for example Codex CLI/Desktop) manage compaction themselves. */
clientManagedResponsesContext?: boolean;
/**
* #10225: request-scoped flag — prompt compression is enabled for this request
* (global compression switch ON and not opted-out by the API key). When set, the
* combo preflight defers its hard context-overflow rejection so chatCore's
* compression runs before the final context gate.
*/
deferContextOverflowWhenCompressible?: boolean;
/** Server-side compression exclusions (#8034) — used to check which targets can run compression. */
compressionExclusions?: CompressionExclusions;
/**
* #10503: request-shape facts (mirroring chatCore.ts's own resolution) threaded
* down to getKnownContextOverflow so the deferral decision can be target-aware —
* a native-Codex-Responses-passthrough target must never count as "compressible"
* (chatCore disables compression for it unconditionally). See
* knownContextOverflow.ts::KnownContextOverflowOptions for the full rationale.
*/
sourceFormat?: string | null;
endpointPath?: string | null;
requestHeaders?: Headers | Record<string, unknown> | null;
};
export type HandleRoundRobinOptions = Omit<HandleComboChatOptions, "apiKeyAllowedConnections">;

View File

@@ -33,6 +33,8 @@ import type { SingleModelTarget } from "@omniroute/open-sse/services/combo/types
import { mergeAbortSignals } from "@omniroute/open-sse/executors/base.ts";
import { resolveRequestAutoControls } from "@omniroute/open-sse/services/autoCombo/requestControls.ts";
import { isVerifiedNativeCodexRequest } from "@omniroute/open-sse/config/codexIdentity.ts";
import { resolveCompressionSettings } from "@omniroute/open-sse/handlers/chatCore/compressionSettings.ts";
import type { CompressionExclusions } from "@omniroute/open-sse/services/compression/exclusions.ts";
import { resolveComboConfig } from "@omniroute/open-sse/services/comboConfig.ts";
import { injectHandoffIntoBody } from "@omniroute/open-sse/services/contextHandoff.ts";
import {
@@ -209,6 +211,31 @@ let combosCacheTs = 0;
let combosCacheVersionSnapshot = -1;
const COMBOS_CACHE_TTL_MS = 10_000;
/**
* #10225 — resolve whether this request's combo preflight should DEFER its hard
* context-overflow rejection so chatCore's compression runs first.
*
* Mirrors handleChatCore's own enablement determination (chatCore.ts): defer only
* when the global compression switch is ON and the API key has not opted out
* (`apiKeyInfo.compressionEnabled !== false`). Per-target applicability (server-side
* exclusions) is checked inside getKnownContextOverflow via the returned exclusions.
* Fail closed (defer=false) on any lookup error — the existing hard preflight stays.
*/
async function resolveComboContextOverflowDeferral(
logger: { warn?: (...args: unknown[]) => void } | null | undefined,
apiKeyInfo: { compressionEnabled?: boolean } | null | undefined
): Promise<{ defer: boolean; exclusions: CompressionExclusions | undefined }> {
try {
const compression = await resolveCompressionSettings(logger);
return {
defer: compression.enabled && apiKeyInfo?.compressionEnabled !== false,
exclusions: compression.settings?.exclusions,
};
} catch {
return { defer: false, exclusions: undefined };
}
}
async function getCombosCachedForChat(): Promise<unknown[]> {
const now = Date.now();
// Explicit non-null check: we intentionally cache and return the Promise
@@ -824,9 +851,20 @@ async function handleChatImplementation(
// Context-relay keeps generation in combo.ts, but handoff injection lives here
// because only this layer knows which connectionId was actually selected.
const { defer: deferContextOverflowWhenCompressible, exclusions: compressionExclusions } =
await resolveComboContextOverflowDeferral(log, apiKeyInfo);
const response = await (handleComboChat as any)({
body,
combo,
deferContextOverflowWhenCompressible,
compressionExclusions,
// #10503: same request-shape facts chatCore.ts resolves for itself
// (resolveChatCoreRequestFormat), so getKnownContextOverflow's target-aware
// deferral check can never drift from chatCore's own native-codex-passthrough
// decision. See knownContextOverflow.ts::KnownContextOverflowOptions.
sourceFormat,
endpointPath: new URL(request.url).pathname,
requestHeaders: request.headers,
clientManagedResponsesContext:
sourceFormat === "openai-responses" &&
new URL(request.url).pathname.split("/").includes("responses") &&
@@ -1103,11 +1141,22 @@ async function handleSingleModelChat(
);
log.info("ROUTING", `Auto-combo redirect from handleSingleModelChat for "${modelStr}"`);
log.info("ROUTING", `Auto-combo redirect to combo flow for "${modelStr}"`);
const { defer: sNetDefer, exclusions: sNetExclusions } =
await resolveComboContextOverflowDeferral(log, apiKeyInfo);
// #10503: same request-shape facts chatCore.ts resolves for itself — threaded
// down so getKnownContextOverflow's target-aware deferral check can never drift
// from chatCore's own native-codex-passthrough decision.
const sNetSourceFormat = detectFormatFromEndpoint(body, clientRawRequest?.endpoint || "");
return handleComboChat({
body,
combo: redirectCombo,
deferContextOverflowWhenCompressible: sNetDefer,
compressionExclusions: sNetExclusions,
sourceFormat: sNetSourceFormat,
endpointPath: clientRawRequest?.endpoint || "",
requestHeaders: clientRawRequest?.headers,
clientManagedResponsesContext:
detectFormatFromEndpoint(body, clientRawRequest?.endpoint || "") === "openai-responses" &&
sNetSourceFormat === "openai-responses" &&
String(clientRawRequest?.endpoint || "")
.split("/")
.includes("responses") &&

View File

@@ -0,0 +1,409 @@
import test from "node:test";
import assert from "node:assert/strict";
import fs from "node:fs";
import os from "node:os";
import path from "node:path";
/**
* #10225 — combo known-context-overflow must NOT hard-reject a compressible
* request before OmniRoute's compression pipeline can run.
*
* Root cause: getKnownContextOverflow() estimates the RAW body (ceil(serializedChars/4)
* over the whole Responses input[]) during combo target resolution, before any
* compression. When every known target limit is below that raw estimate, both call
* sites (round-robin + target-resolution) convert it into an immediate local 400
* `context_length_exceeded` with attempted:0 — so chatCore's proactive compression
* (which can shrink 294133→111529, 62% in the reporter's case) never runs. The only
* existing bypass (clientManagedResponsesContext) is gated to VERIFIED native Codex
* clients, so a generic Responses client (e.g. OpenCode) pointed at a codex model
* still hits the hard gate.
*
* Fix: thread a request-scoped `deferContextOverflowWhenCompressible` flag (set when
* the global compression switch is ON and not API-key opted-out). When set AND at
* least one target can run compression, getKnownContextOverflow returns null so the
* request reaches chatCore, whose post-compression enforceOutputTokenBudget becomes
* the final context gate — a local 400 only if the compressed body still cannot fit.
*/
const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-combo-overflow-compress-"));
const ORIGINAL_DATA_DIR = process.env.DATA_DIR;
process.env.DATA_DIR = TEST_DATA_DIR;
const core = await import("../../src/lib/db/core.ts");
const { saveModelsDevCapabilities, clearModelsDevCapabilities } =
await import("../../src/lib/modelsDevSync.ts");
const { getKnownContextOverflow, handleComboChat } = await import(
"../../open-sse/services/combo.ts"
);
const { updateCompressionSettings } = await import("../../src/lib/db/compression.ts");
const { handleChatCore } = await import("../../open-sse/handlers/chatCore.ts");
test.after(() => {
core.resetDbInstance();
if (ORIGINAL_DATA_DIR === undefined) {
delete process.env.DATA_DIR;
} else {
process.env.DATA_DIR = ORIGINAL_DATA_DIR;
}
fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true });
});
test.beforeEach(() => {
clearModelsDevCapabilities();
});
function capabilityEntry(limitContext: number | null) {
return {
tool_call: true,
reasoning: false,
attachment: false,
structured_output: true,
temperature: true,
modalities_input: JSON.stringify(["text"]),
modalities_output: JSON.stringify(["text"]),
knowledge_cutoff: null,
release_date: null,
last_updated: null,
status: null,
family: null,
open_weights: false,
limit_context: limitContext,
limit_input: limitContext,
limit_output: 4096,
interleaved_field: null,
};
}
function target(modelStr: string) {
return {
kind: "model" as const,
stepId: modelStr,
executionKey: modelStr,
modelStr,
provider: modelStr.includes("/") ? modelStr.split("/")[0] : modelStr,
providerId: null,
connectionId: null,
weight: 1,
label: null,
};
}
// A generic Responses-API body whose estimate lands near `tokens` tokens (4 chars/token).
// Uses `input:` (not `messages:`) to mirror the OpenCode/Codex Responses surface.
function bigResponsesBody(tokens: number) {
return { input: [["user", "x".repeat(tokens * 4)]] };
}
const noopLog = { info() {}, warn() {}, error() {}, debug() {} };
test("#10225 getKnownContextOverflow defers the hard overflow when compression is available", () => {
saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } });
const body = bigResponsesBody(275_000);
// Compression enabled + target can compress -> defer (null).
assert.equal(
getKnownContextOverflow([target("codex/gpt-5.6-terra")], body, {
deferContextOverflowWhenCompressible: true,
}),
null,
"compressible request must defer so chatCore compression can run (#10225)"
);
// Compression disabled -> the existing hard overflow is preserved (never lose #7177).
const hard = getKnownContextOverflow([target("codex/gpt-5.6-terra")], body);
assert.ok(hard);
assert.ok(hard.requiredContextTokens > hard.maxKnownContextTokens);
// Compression enabled but EVERY target is excluded from compression -> keep the hard gate.
const excluded = getKnownContextOverflow([target("codex/gpt-5.6-terra")], body, {
deferContextOverflowWhenCompressible: true,
compressionExclusions: ["gpt-5.6-terra"],
});
assert.ok(excluded, "fully-excluded targets must retain the hard preflight");
});
test("#10225 combo does not early-400 a compressible over-limit request when deferral is on", async () => {
saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } });
let dispatches = 0;
const response = await handleComboChat({
body: bigResponsesBody(275_000),
combo: {
name: "codex-compress-overflow",
strategy: "priority",
models: ["codex/gpt-5.6-terra"],
},
deferContextOverflowWhenCompressible: true,
clientManagedResponsesContext: false,
isModelAvailable: async () => true,
handleSingleModel: async () => {
dispatches += 1;
return new Response("ok", { status: 200 });
},
log: noopLog,
});
assert.notEqual(response.status, 400, "compression-enabled request must reach chatCore");
assert.equal(dispatches, 1, "must dispatch so chatCore compaction runs first");
});
test("#10225 combo keeps the fast 400 when compression is disabled", async () => {
saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } });
let dispatches = 0;
const response = await handleComboChat({
body: bigResponsesBody(275_000),
combo: {
name: "codex-compress-disabled",
strategy: "priority",
models: ["codex/gpt-5.6-terra"],
},
deferContextOverflowWhenCompressible: false,
clientManagedResponsesContext: false,
isModelAvailable: async () => true,
handleSingleModel: async () => {
dispatches += 1;
return new Response("ok", { status: 200 });
},
log: noopLog,
});
assert.equal(response.status, 400);
assert.equal(dispatches, 0, "#7177 anti-exhaustion guard must survive when compression is off");
const body = await response.json();
assert.equal(body.error.code, "context_length_exceeded");
});
// #10501-sweep #10503 — the deferral above is NOT target-aware by default: it only
// checks operator-named compression exclusions, never whether chatCore will actually
// attempt compression for the resolved target. handleChatCore.ts unconditionally sets
// `compressionExcluded = nativeCodexPassthrough || ...` for a verified native Codex
// Responses passthrough target (open-sse/handlers/chatCore.ts) — deferring the
// preflight there means an oversized request sails past BOTH gates uncompressed. These
// tests pin the fix: a native-codex-passthrough target must never count toward "can
// compress", so the hard preflight stays active and no upstream dispatch happens.
// NOTE on `clientManagedResponsesContext: false` below: these tests deliberately do
// NOT set it, to isolate the fix from the PRE-EXISTING, unrelated early-return a few
// lines above in knownContextOverflow.ts ("Native Codex Responses clients compact
// their own item history") — that block ALSO returns null for an all-codex pool, but
// only when `clientManagedResponsesContext === true` (a VERIFIED native client). The
// bug this fix targets is broader: chatCore's `shouldUseNativeCodexPassthrough` short-
// circuits to true for `provider === "codex"` regardless of verification (see
// passthroughHelpers.ts), so an UNVERIFIED request that nonetheless targets a `codex`
// combo member over `/v1/responses` in openai-responses format still hits chatCore's
// compression bypass — exactly the gap `sourceFormat`/`endpointPath` (not the looser
// `clientManagedResponsesContext` flag) now closes.
test("#10503 getKnownContextOverflow REFUSES to defer when the only target is native Codex Responses passthrough", () => {
saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } });
const body = bigResponsesBody(275_000);
const overflow = getKnownContextOverflow([target("codex/gpt-5.6-terra")], body, {
deferContextOverflowWhenCompressible: true,
sourceFormat: "openai-responses",
endpointPath: "/v1/responses",
});
assert.ok(
overflow,
"a native-codex-passthrough target must never be treated as compressible — the " +
"hard preflight must stay active (chatCore disables compression for it entirely)"
);
});
test("#10503 getKnownContextOverflow still defers when a genuinely compressible sibling target is present", () => {
saveModelsDevCapabilities({
codex: { "gpt-5.6-terra": capabilityEntry(272_000) },
openai: { "gpt-5.6-terra": capabilityEntry(272_000) },
});
const body = bigResponsesBody(275_000);
// A heterogeneous pool where at least ONE target (openai) genuinely runs
// compression must still defer — deferral is a per-request decision, and other
// targets in the pool are unaffected by the codex-specific compression bypass.
const overflow = getKnownContextOverflow(
[target("codex/gpt-5.6-terra"), target("openai/gpt-5.6-terra")],
body,
{
deferContextOverflowWhenCompressible: true,
sourceFormat: "openai-responses",
endpointPath: "/v1/responses",
}
);
assert.equal(overflow, null, "a genuinely compressible sibling target must still defer");
});
test("#10503 handleComboChat: native-codex-passthrough pool fails FAST locally, zero upstream dispatches", async () => {
saveModelsDevCapabilities({ codex: { "gpt-5.6-terra": capabilityEntry(272_000) } });
let dispatches = 0;
const response = await handleComboChat({
body: bigResponsesBody(275_000),
combo: {
name: "codex-native-passthrough-overflow",
strategy: "priority",
models: ["codex/gpt-5.6-terra"],
},
deferContextOverflowWhenCompressible: true,
sourceFormat: "openai-responses",
endpointPath: "/v1/responses",
isModelAvailable: async () => true,
handleSingleModel: async () => {
dispatches += 1;
return new Response("ok", { status: 200 });
},
log: noopLog,
});
assert.equal(
response.status,
400,
"must fail fast locally instead of dispatching an oversized, uncompressible request"
);
assert.equal(dispatches, 0, "no wasted upstream call for a target that can never compress");
const responseBody = await response.json();
assert.equal(responseBody.error.code, "context_length_exceeded");
});
// #10503 item 2 — drive the REAL chatCore compression pipeline end-to-end (not just the
// pure getKnownContextOverflow helper): a genuinely compressible multi-turn request must
// have chatCore's proactive/last-resort compression actually run and dispatch the
// COMPRESSED body upstream; a request that is STILL too large after compression must be
// rejected locally with zero upstream dispatch (fail-fast, matching the codex-passthrough
// case above in outcome, but via the "compression tried and wasn't enough" path instead
// of "compression was never eligible").
//
// Uses an unregistered synthetic provider + CONTEXT_LENGTH_<PROVIDER> env override
// (same technique as tests/unit/chatcore-combo-context-limit-8378.test.ts) so the
// context limit is small and deterministic without depending on any real catalog entry.
// Compression targets conversation HISTORY (older turns), not the current terminal
// message — this is why the fixtures below build many small history turns plus one
// short final turn (compressible case) vs one large, irreducible final turn
// (still-too-large case).
const CHATCORE_PROBE_PROVIDER = "combo10503probe";
const CHATCORE_PROBE_MODEL = "combo10503probemodel";
const CHATCORE_LIMIT_ENV = "CONTEXT_LENGTH_COMBO10503PROBE";
function buildHistoryBody(turns: number, finalMessageChars: number) {
const messages: Array<{ role: string; content: string }> = [
{ role: "system", content: "You are a helpful assistant." },
];
for (let i = 0; i < turns; i++) {
messages.push({
role: "user",
content: `Message number ${i}: Alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima mno pqr stu.`,
});
messages.push({
role: "assistant",
content: `Reply number ${i}: verbose filler answer with extra padding text for realism.`,
});
}
messages.push({ role: "user", content: "FINAL: " + "z".repeat(finalMessageChars) });
return { model: CHATCORE_PROBE_MODEL, messages, stream: false };
}
async function invokeChatCoreCapturingUpstream(body: Record<string, unknown>) {
const originalFetch = globalThis.fetch;
let dispatched = false;
let sentBodyJson: string | null = null;
globalThis.fetch = async (_url: RequestInfo | URL, init: RequestInit = {}) => {
dispatched = true;
sentBodyJson = init.body ? String(init.body) : null;
return new Response(
JSON.stringify({
id: "chatcmpl-10503",
object: "chat.completion",
model: CHATCORE_PROBE_MODEL,
choices: [
{ index: 0, message: { role: "assistant", content: "ok" }, finish_reason: "stop" },
],
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
}),
{ status: 200, headers: { "content-type": "application/json" } }
);
};
try {
const result = await handleChatCore({
body,
modelInfo: {
provider: CHATCORE_PROBE_PROVIDER,
model: CHATCORE_PROBE_MODEL,
extendedContext: false,
},
credentials: { apiKey: "sk-test", providerSpecificData: {} },
log: { debug() {}, info() {}, warn() {}, error() {} },
clientRawRequest: {
endpoint: "/v1/chat/completions",
body,
headers: new Headers({ accept: "application/json" }),
},
userAgent: "unit-test",
} as never);
return { result, dispatched, sentBodyJson };
} finally {
globalThis.fetch = originalFetch;
}
}
test("#10503 real chatCore path: a compressible request dispatches the COMPRESSED (not raw) body upstream", async () => {
const originalEnv = process.env[CHATCORE_LIMIT_ENV];
process.env[CHATCORE_LIMIT_ENV] = "500";
await updateCompressionSettings({
enabled: true,
defaultMode: "standard",
autoTriggerTokens: 1,
autoTriggerMode: "standard",
engines: { rtk: { enabled: true }, caveman: { enabled: true } },
} as never);
try {
const body = buildHistoryBody(150, 50);
const rawLen = JSON.stringify(body.messages).length;
const { dispatched, sentBodyJson } = await invokeChatCoreCapturingUpstream(body);
assert.ok(dispatched, "compression must let a genuinely compressible request reach chatCore's dispatch");
assert.ok(sentBodyJson, "the dispatched request must carry a body");
assert.ok(
sentBodyJson!.length < rawLen * 0.5,
`expected the DISPATCHED body (${sentBodyJson!.length} chars) to be substantially ` +
`smaller than the raw request (${rawLen} chars) — proves compression actually ran ` +
`and its output (not the raw body) is what reached upstream`
);
} finally {
if (originalEnv === undefined) delete process.env[CHATCORE_LIMIT_ENV];
else process.env[CHATCORE_LIMIT_ENV] = originalEnv;
}
});
test("#10503 real chatCore path: STILL too large after compression → local rejection, ZERO upstream dispatch", async () => {
const originalEnv = process.env[CHATCORE_LIMIT_ENV];
process.env[CHATCORE_LIMIT_ENV] = "50";
await updateCompressionSettings({
enabled: true,
defaultMode: "standard",
autoTriggerTokens: 1,
autoTriggerMode: "standard",
engines: { rtk: { enabled: true }, caveman: { enabled: true } },
} as never);
try {
// The final turn alone (1000 chars, irreducible — compression trims HISTORY, not
// the current terminal message) already exceeds the 50-token limit, so no amount
// of history compaction can make this fit.
const body = buildHistoryBody(150, 1000);
const { result, dispatched } = await invokeChatCoreCapturingUpstream(body);
assert.equal(
dispatched,
false,
"fail-fast: a request that cannot fit even after compression must never reach fetch()"
);
assert.equal((result as { success: boolean }).success, false);
const failure = result as { success: false; error?: string; rawMessage?: string };
const message = failure.rawMessage ?? failure.error ?? "";
assert.match(message, /exceeds/i);
} finally {
if (originalEnv === undefined) delete process.env[CHATCORE_LIMIT_ENV];
else process.env[CHATCORE_LIMIT_ENV] = originalEnv;
}
});