Files
OmniRoute/open-sse/services/combo/contextOverrideGate.ts
Bob.Hou e8a91173de fix(combos): inherit model_context_overrides onto effort-suffixed targets (#12475) (#12926)
Validado numa worktree combinada com as 16 PRs desta leva sobre `release/v3.8.51`: typecheck:core limpo, check-file-size e check-changelog-integrity OK, complexity 2788/3218 e cognitive 1261/1437 (ambos sob a baseline), ESLint 0 erros nos 152 arquivos alterados, 771 testes unitários focados, 49 de integração e a suíte vitest:ui completa (2149) verdes.

A busca do sufixo mais longo primeiro (`-xhigh` antes de `-high`) é o detalhe que faz a herança funcionar em vez de quase-funcionar. Manter `getResolvedModelContextOverride` fora do escopo, com o teste existente registrando que aquele caminho continua sem herança, deixa a fronteira explícita.
2026-09-07 08:57:27 -03:00

143 lines
5.8 KiB
TypeScript

/**
* Context-fit evaluation for combo routing's compatibility filter, extracted
* from comboStructure.ts to keep that file under the file-size cap (PR
* #7933's model_context_override fix pushed it over).
*
* evaluateContextLimit() is the single chokepoint both compatibility-check
* call sites in comboStructure.ts (hasKnownCompatibleContextLimit,
* getTargetCompatibilityFailures) go through. It first consults a persisted
* per-model context override, then falls back to the catalog's
* maxInputTokens/contextWindow limits.
*
* Override rationale (Feature 5004): the catalog's `maxInputTokens` can be a
* deliberately smaller *client-facing* hint (e.g. set below the true window so
* coding agents auto-compact — #6191); using it to filter fallback targets
* wrongly drops otherwise-capable providers for large prompts, collapsing the
* pool to one provider and producing a hard 503 with no fallback once that
* provider's quota is exhausted. An operator-set or auto-discovered override
* reflects the real capacity, so it supersedes both catalog limits. Uses the
* resolved exact override (`getResolvedModelContextOverride` returns `null` when none is set) —
* NOT `getModelContextLimitForModelString`, which falls back to
* `contextWindow` and would therefore bypass the `maxInputTokens` cap for
* every model, not just overridden ones.
*/
import { getResolvedModelContextOverride } from "../../../src/lib/modelCapabilities";
import { parseModel } from "../model.ts";
/**
* Longest-first so `-xhigh` is not eaten by `-high`. Mirrors
* `stripKnownEffortSuffix` in modelCapabilities.ts, but that helper's array
* order still matches `-high` first (`"…-xhigh".endsWith("-high")`).
*/
const EFFORT_SUFFIXES_LONGEST_FIRST = [
"minimal",
"medium",
"xhigh",
"none",
"high",
"max",
"low",
] as const;
function stripTrailingEffortSuffix(modelId: string): string | null {
const normalized = String(modelId || "").trim();
if (!normalized) return null;
const lowered = normalized.toLowerCase();
for (const suffix of EFFORT_SUFFIXES_LONGEST_FIRST) {
const token = `-${suffix}`;
if (lowered.length > token.length && lowered.endsWith(token)) {
return normalized.slice(0, -token.length);
}
}
return null;
}
/**
* Exact override first; if missing, inherit the base id after stripping a
* trailing effort tier (#12475). Combo members are stored as
* `provider/GLM-5.3-high` while `model_context_overrides` is keyed on
* `GLM-5.3`. Dispatcher already strips the suffix; the compat filter did not.
*/
function lookupOverrideWithEffortInheritance(modelStr: string): number | null {
const exact = getResolvedModelContextOverride(modelStr);
if (exact != null) return exact;
const parsed = parseModel(modelStr);
const modelId = typeof parsed.model === "string" ? parsed.model.trim() : "";
const base = stripTrailingEffortSuffix(modelId);
if (!base || base === modelId) return null;
if (parsed.provider) {
return getResolvedModelContextOverride({ provider: parsed.provider, model: base });
}
return getResolvedModelContextOverride(base);
}
/**
* Resolve the context-fit verdict from a persisted per-model override, if one
* is set. Returns `undefined` when there is no `modelStr` or no override
* exists, so the caller falls through to the catalog-based check; otherwise
* returns the fit verdict for the override itself.
*/
function resolveContextOverrideVerdict(
modelStr: string | undefined,
requiredContextTokens: number
): boolean | undefined {
if (!modelStr) return undefined;
const override = lookupOverrideWithEffortInheritance(modelStr);
if (override == null) return undefined;
return override >= requiredContextTokens;
}
/**
* Decide whether a target's known context limit accommodates the request.
*
* `maxInputTokens` is an **input-only** cap — the requested output reserve is
* already enforced separately against `maxOutputTokens` (see
* `exceedsKnownOutputLimit` in comboStructure.ts), so it must NOT be
* re-counted here. Comparing `maxInputTokens` against `estimatedInputTokens +
* requestedOutputTokens` double-counted the output reserve and shrank the
* effective input allowance (#7039).
*
* `contextWindow` is the total window, so input + output must both fit.
*
* Returns `true` when the known limit accommodates the request, `false` when
* it is known to be too small, and `null` when no limit metadata is known.
*/
export function evaluateContextLimit(
capabilities: { maxInputTokens?: number | null; contextWindow?: number | null },
requirements: { estimatedInputTokens: number; requiredContextTokens: number },
modelStr?: string
): boolean | null {
const overrideVerdict = resolveContextOverrideVerdict(
modelStr,
requirements.requiredContextTokens
);
if (overrideVerdict !== undefined) return overrideVerdict;
const hasMaxInput = capabilities.maxInputTokens != null;
const hasContextWindow = capabilities.contextWindow != null;
// Neither limit is known — cannot judge.
if (!hasMaxInput && !hasContextWindow) return null;
// The input-only cap must accommodate the estimated input.
const inputFits = hasMaxInput
? capabilities.maxInputTokens! >= requirements.estimatedInputTokens
: true;
// The total window must accommodate input + requested output. The output
// reserve is enforced separately via `maxOutputTokens`, but when a model
// exposes both `maxInputTokens` and `contextWindow` the two must not be
// checked in isolation: a request whose input fits `maxInputTokens` but whose
// input + output exceeds `contextWindow` must still be rejected (#7039
// follow-up — shared-window models where `maxInputTokens` defaults to the
// total window size).
const totalFits = hasContextWindow
? capabilities.contextWindow! >= requirements.requiredContextTokens
: true;
return inputFits && totalFits;
}