From 2355f7beb3b90671abcb83f0758366ff9ffe9a73 Mon Sep 17 00:00:00 2001 From: James Date: Thu, 13 Aug 2026 19:53:52 +0900 Subject: [PATCH] fix(translator): resolve the Claude thinking output cap with the routed provider (#10139) (#10238) fitThinkingToMaxTokens() clamps the synthesized max_tokens to the model output cap, but resolved that cap from a bare model id via safeCapMaxOutputTokens(model) -> capMaxOutputTokens(model). A cap that is only known per provider -- an operator max_output_tokens override, a synced catalog limit_output, or a registry entry -- is invisible to a bare-model lookup, so modelCap came back null and the unbounded responseRoom + requestedBudget branch ran. When the client sends no max-token field at all, adjustMaxTokens() supplies DEFAULT_MAX_TOKENS (64000) and reasoning_effort: "high" supplies a 131072 thinking budget, so the provider request carried max_tokens: 195072 and every such request was rejected upstream with a bare 400. Thread the already-in-scope routedProvider (openai-to-claude.ts:122, used two lines later for the Kimi-coding check) through fitThinkingToMaxTokens() into capMaxOutputTokens({ provider, model }), which already supports provider-scoped resolution via resolveCapabilityInput() -- no new lookup path needed. Omitting the provider (existing callers, tests) keeps the bare-model behavior unchanged; verified in the added regression test. Follow-up to #6637, whose token-budgeting half was never addressed: #6893 fixed only the combo fallback classification. Rebased onto the open-sse/translator/request/openai-to-claude/thinkingBudget.ts extraction that landed after the original patch was written against the inline code in openai-to-claude.ts. --- ...0139-thinking-output-cap-provider-scope.md | 1 + .../translator/request/openai-to-claude.ts | 7 +- .../openai-to-claude/thinkingBudget.ts | 15 ++- ...o-10139-claude-thinking-output-cap.test.ts | 110 ++++++++++++++++++ 4 files changed, 128 insertions(+), 5 deletions(-) create mode 100644 changelog.d/fixes/10139-thinking-output-cap-provider-scope.md create mode 100644 tests/unit/repro-10139-claude-thinking-output-cap.test.ts diff --git a/changelog.d/fixes/10139-thinking-output-cap-provider-scope.md b/changelog.d/fixes/10139-thinking-output-cap-provider-scope.md new file mode 100644 index 0000000000..6fc97e467a --- /dev/null +++ b/changelog.d/fixes/10139-thinking-output-cap-provider-scope.md @@ -0,0 +1 @@ +- **fix(translator):** resolve the Claude thinking output cap with the routed provider so a provider-scoped-only `max_output_tokens` override is no longer invisible to `fitThinkingToMaxTokens()`, which previously let the synthesized `max_tokens` (caller room + thinking budget) go out unbounded and 400 upstream ([#10139](https://github.com/diegosouzapw/OmniRoute/issues/10139)) diff --git a/open-sse/translator/request/openai-to-claude.ts b/open-sse/translator/request/openai-to-claude.ts index cc0643914e..7fb844bda9 100644 --- a/open-sse/translator/request/openai-to-claude.ts +++ b/open-sse/translator/request/openai-to-claude.ts @@ -240,7 +240,12 @@ export function openaiToClaudeRequest(model, body, stream, credentials = null) { // could exceed model caps (e.g. Opus 4.7's 128000 ceiling) and trigger // HTTP 400 from Anthropic. if (!isKimiCoding) { - const fitted = fitThinkingToMaxTokens(model, Number(result.max_tokens) || 0, result.thinking); + const fitted = fitThinkingToMaxTokens( + model, + Number(result.max_tokens) || 0, + result.thinking, + routedProvider + ); result.max_tokens = fitted.maxTokens; if (fitted.thinking === undefined) { delete result.thinking; diff --git a/open-sse/translator/request/openai-to-claude/thinkingBudget.ts b/open-sse/translator/request/openai-to-claude/thinkingBudget.ts index e78275570f..fa2333e059 100644 --- a/open-sse/translator/request/openai-to-claude/thinkingBudget.ts +++ b/open-sse/translator/request/openai-to-claude/thinkingBudget.ts @@ -7,9 +7,9 @@ import { capMaxOutputTokens } from "../../../../src/lib/modelCapabilities.ts"; const MIN_CLAUDE_THINKING_BUDGET = 1024; const MIN_RESPONSE_ROOM = 1024; -function safeCapMaxOutputTokens(model: string): number | null { +function safeCapMaxOutputTokens(model: string, provider?: string | null): number | null { try { - const cap = capMaxOutputTokens(model); + const cap = capMaxOutputTokens(provider ? { provider, model } : model); return typeof cap === "number" && cap > 0 ? cap : null; } catch { return null; @@ -31,6 +31,12 @@ function safeCapMaxOutputTokens(model: string): number | null { * responseRoom shrunk to MIN_RESPONSE_ROOM; if still below MIN, disable * thinking entirely (cap too tight for any reasoning). * + * `provider` scopes the cap lookup to a provider-specific override (e.g. a + * dashboard-set `max_output_tokens` for `opencode-go/qwen3.7-plus`) when the + * model-only entry has no cap of its own. Without it, a model whose real + * ceiling is only known per-provider resolves to no cap at all and the + * synthesized `max_tokens` goes out unbounded (#10139). + * * Worked example (real-world Opus 4.7 case that previously 400'd): * caller max_tokens = 32000, reasoning_effort=high → budget = 131072, * model cap = 128000. @@ -42,9 +48,10 @@ function safeCapMaxOutputTokens(model: string): number | null { export function fitThinkingToMaxTokens( model: string, callerMaxTokens: number, - thinking: Record | undefined + thinking: Record | undefined, + provider?: string | null ): { maxTokens: number; thinking: Record | undefined } { - const modelCap = safeCapMaxOutputTokens(model); + const modelCap = safeCapMaxOutputTokens(model, provider); const requestedBudget = Number(thinking?.budget_tokens) || 0; // No budgeted thinking — just cap max_tokens to the model output ceiling. diff --git a/tests/unit/repro-10139-claude-thinking-output-cap.test.ts b/tests/unit/repro-10139-claude-thinking-output-cap.test.ts new file mode 100644 index 0000000000..26661a518b --- /dev/null +++ b/tests/unit/repro-10139-claude-thinking-output-cap.test.ts @@ -0,0 +1,110 @@ +/** + * Regression test for issue #10139 (follow-up to #6637). + * + * Reporter observed: for a model whose real output ceiling is only known via a + * provider-scoped override (e.g. an operator-set `max_output_tokens` for + * `opencode-go/qwen3.7-plus`, written through the dashboard's + * `PUT /api/provider-models`), `fitThinkingToMaxTokens()` resolved the cap by + * model name alone. The provider-scoped override was invisible to that lookup, + * so the cap resolved to `null` and the synthesized `max_tokens` (caller's + * `max_tokens` + the `reasoning_effort` thinking budget) went out unbounded — + * `64000 + 131072 = 195072`, which the upstream provider rejected with a bare + * `400 {"model":"qwen3.7-plus"}`. + * + * Root cause (confirmed by reading `fitThinkingToMaxTokens` and its caller in + * `openai-to-claude.ts`): the caller already has `routedProvider` in scope + * (used two lines earlier for the Kimi-coding check) but never passed it + * through, so `safeCapMaxOutputTokens()` always resolved the model-only + * capability entry and missed any provider-scoped override. + * + * Fix: `fitThinkingToMaxTokens()` takes an optional `provider` argument and + * forwards it to `capMaxOutputTokens({ provider, model })`, which already + * supports provider-scoped resolution; the caller now passes `routedProvider`. + */ +import assert from "node:assert/strict"; +import test from "node:test"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; + +const TEST_DATA_DIR = fs.mkdtempSync(path.join(os.tmpdir(), "omniroute-repro-10139-")); +process.env.DATA_DIR = TEST_DATA_DIR; + +const core = await import("../../src/lib/db/core.ts"); +const { setModelCapabilityOverride, removeModelCapabilityOverride } = + await import("../../src/lib/db/modelCapabilityOverrides.ts"); +const { fitThinkingToMaxTokens } = + await import("../../open-sse/translator/request/openai-to-claude/thinkingBudget.ts"); + +const PROVIDER = "opencode-go"; +const MODEL = "qwen3.7-plus"; +const TARGET = `${PROVIDER}/${MODEL}`; +const REAL_UPSTREAM_OUTPUT_CAP = 131072; // per reporter's boundary test (131072 -> 200, 131073 -> 400) +const CALLER_MAX_TOKENS = 64000; // DEFAULT_MAX_TOKENS when the client sends none +const THINKING_BUDGET = 131072; // effortBudgetMap.high + +test.after(() => { + core.resetDbInstance(); + fs.rmSync(TEST_DATA_DIR, { recursive: true, force: true }); +}); + +test("#10139: a provider-scoped-only output cap is invisible without a provider argument", () => { + assert.ok( + setModelCapabilityOverride(TARGET, "max_output_tokens", REAL_UPSTREAM_OUTPUT_CAP), + "expected the max_output_tokens override to be written" + ); + try { + // Reproduces the pre-fix call shape: no provider passed. + const result = fitThinkingToMaxTokens(MODEL, CALLER_MAX_TOKENS, { + type: "enabled", + budget_tokens: THINKING_BUDGET, + }); + // This is the defect: with no provider, the override is invisible and the + // synthesized max_tokens goes out unbounded (64000 + 131072 = 195072). + assert.equal( + result.maxTokens, + CALLER_MAX_TOKENS + THINKING_BUDGET, + "documents the pre-fix behavior: uncapped when provider is omitted" + ); + } finally { + removeModelCapabilityOverride(TARGET, "max_output_tokens"); + } +}); + +test("#10139: passing the routed provider resolves the override and clamps max_tokens", () => { + assert.ok( + setModelCapabilityOverride(TARGET, "max_output_tokens", REAL_UPSTREAM_OUTPUT_CAP), + "expected the max_output_tokens override to be written" + ); + try { + const result = fitThinkingToMaxTokens( + MODEL, + CALLER_MAX_TOKENS, + { type: "enabled", budget_tokens: THINKING_BUDGET }, + PROVIDER + ); + assert.ok( + result.maxTokens <= REAL_UPSTREAM_OUTPUT_CAP, + `expected max_tokens to stay <= ${REAL_UPSTREAM_OUTPUT_CAP}, got ${result.maxTokens} ` + + `(reproduces the reported 195072, upstream then 400s)` + ); + assert.equal(result.maxTokens, REAL_UPSTREAM_OUTPUT_CAP); + assert.equal(result.thinking?.budget_tokens, REAL_UPSTREAM_OUTPUT_CAP - CALLER_MAX_TOKENS); + } finally { + removeModelCapabilityOverride(TARGET, "max_output_tokens"); + } +}); + +test("#10139: no override present, provider argument is a no-op (unchanged behavior)", () => { + const withoutProvider = fitThinkingToMaxTokens(MODEL, CALLER_MAX_TOKENS, { + type: "enabled", + budget_tokens: THINKING_BUDGET, + }); + const withProvider = fitThinkingToMaxTokens( + MODEL, + CALLER_MAX_TOKENS, + { type: "enabled", budget_tokens: THINKING_BUDGET }, + PROVIDER + ); + assert.deepEqual(withProvider, withoutProvider); +});