fix(executors): granular reasoning_effort handling for Claude models on Copilot (#4443)

Integrated into release/v3.8.32
This commit is contained in:
Diego Rodrigues de Sa e Souza
2026-06-20 20:25:39 -03:00
committed by GitHub
parent 53a3b597b2
commit 0403f29805
4 changed files with 165 additions and 3 deletions

View File

@@ -18,6 +18,7 @@ _In development — bullets added per PR; finalized at release._
- **fix(translator):** sanitize Read tool args from non-Anthropic models to prevent retry loops. (thanks @GodrezJr2)
- **fix(usage):** reuse Gemini CLI project ID for quota checks (avoid re-discovery). (thanks @Delcado19)
- **fix(dashboard):** surface manual config CTA when Claude CLI detection fails (remote deployments). (thanks @anuragg-saxenaa)
- **fix(executors):** granular reasoning_effort handling for Claude models on GitHub Copilot. (thanks @baslr)
- **fix(combo): round-robin members fail over faster under concurrency saturation via a configurable queue depth** — when a round-robin combo member was saturated, requests sat in the per-model semaphore's **unbounded** queue and only failed over to the next member after the full `queueTimeoutMs` (default 30s) elapsed — so a burst of agentic requests deep-queued one hot member instead of spilling to healthy ones. The per-model semaphore now accepts a bounded queue depth and emits `SEMAPHORE_QUEUE_FULL` once it is full (the round-robin loop already cascades on that code), so a configured low depth fails over immediately. A new `queueDepth` combo-config knob (global default / provider override / per-combo, default **20** for backward compatibility; **0** = never queue → fail over now) is exposed in Settings → Combo Defaults. ([#3872](https://github.com/diegosouzapw/OmniRoute/issues/3872) — thanks @KooshaPari)
- **fix(pricing): align Claude Code (`cc`) pricing with current Anthropic per-MTok rates** — the `cc` provider block in the default pricing table had stale numbers across every Claude 4.x family entry — most visibly, `claude-opus-4-5-20251101` was billed at the deprecated Opus 4.1 rate (`input $15` / `output $75`), and `claude-haiku-4-5-20251001` was at half the current Haiku 4.5 rate. The `cached` (cache hit) and `cache_creation` (5-minute cache write) multipliers were also off across Opus 4.6/4.7/4.8, Sonnet 4.5/4.6, Haiku 4.5, and Fable 5. All eight entries now match the rates Anthropic publishes (input, 5m cache write at 1.25x input, cache hit at 0.1x input, output; reasoning billed at the output rate), so cost accounting on the dashboard and per-request usage events stop under- or over-reporting Claude Code spend. (thanks @chulanpro5)

View File

@@ -242,6 +242,18 @@ function hasActiveClaudeThinking(body: Record<string, unknown>): boolean {
* xhigh by default and falls back to high only for explicit xhigh opt-outs.
*/
const MISTRAL_NO_REASONING_EFFORT_PATTERN = /devstral/i;
// GitHub Copilot Claude routing is granular (upstream port: decolua/9router#791):
// ✅ Pass through — Claude Opus 4.6, Claude Sonnet 4.6. Copilot routes both to
// Anthropic's chat/completions surface, which honors reasoning_effort and
// emits visible reasoning tokens (verified upstream: 3× token increase
// between low/medium/high).
// ❌ Strip — Claude Haiku 4.5 and Claude Opus 4.7 (rejected upstream by
// Copilot's Claude backend), older Claude variants, all `haiku`-named
// models, and the `oswe-*` family (Raptor) which still rejects
// reasoning_effort.
// Order matters: the opt-in check must run BEFORE the broad Claude/haiku/oswe strip.
const GITHUB_REASONING_EFFORT_OPT_IN_PATTERN =
/claude[-_.]?(?:opus|sonnet)[-_.]?4[-_.]6/i;
const GITHUB_NO_REASONING_EFFORT_PATTERN = /(claude|haiku|oswe)/i;
function supportsMaxEffortForProvider(provider: string, model: string): boolean {
@@ -269,9 +281,11 @@ export function sanitizeReasoningEffortForProvider(
const effortStr = typeof effort === "string" ? effort.toLowerCase() : "";
const modelStr = model || "";
const githubOptIn =
provider === "github" && GITHUB_REASONING_EFFORT_OPT_IN_PATTERN.test(modelStr);
const rejecting =
(provider === "mistral" && MISTRAL_NO_REASONING_EFFORT_PATTERN.test(modelStr)) ||
(provider === "github" && GITHUB_NO_REASONING_EFFORT_PATTERN.test(modelStr));
(provider === "github" && !githubOptIn && GITHUB_NO_REASONING_EFFORT_PATTERN.test(modelStr));
if (rejecting) {
log?.info?.(
"REASONING_SANITIZE",

View File

@@ -250,13 +250,25 @@ test("sanitizeReasoningEffortForProvider: mistral/devstral strips reasoning_effo
);
});
test("sanitizeReasoningEffortForProvider: github/claude-opus strips reasoning_effort entirely", () => {
test("sanitizeReasoningEffortForProvider: github/claude-opus-4.6 preserves reasoning_effort (#791)", () => {
// Upstream PR decolua/9router#791 (port): Copilot now honors reasoning_effort
// on Claude Opus 4.6 and Sonnet 4.6. Older Opus variants and Haiku still strip.
const body = {
model: "claude-opus-4-6",
reasoning_effort: "high",
messages: [],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-opus-4-6", null);
assert.equal((result as any).reasoning_effort, "high");
});
test("sanitizeReasoningEffortForProvider: github/claude-opus-4.7 still strips (#791)", () => {
const body = {
model: "claude-opus-4.7",
reasoning_effort: "high",
messages: [],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-opus-4.7", null);
assert.equal((result as any).reasoning_effort, undefined);
});
@@ -274,6 +286,10 @@ test("sanitizeReasoningEffortForProvider: rejecting providers strip max before n
);
assert.equal((mistralResult as any).reasoning_effort, undefined);
// Pre-#791: github stripped reasoning_effort entirely for every Claude model.
// Post-#791: Opus 4.6 keeps reasoning_effort; `max` downgrades to `high`
// because github is not Claude/CC-compatible (so supportsMax=false) and
// the canonical Claude Opus 4.6 model opts out of xhigh.
const githubBody = {
model: "claude-opus-4-6",
reasoning_effort: "max",
@@ -285,7 +301,22 @@ test("sanitizeReasoningEffortForProvider: rejecting providers strip max before n
"claude-opus-4-6",
null
);
assert.equal((githubResult as any).reasoning_effort, undefined);
assert.equal((githubResult as any).reasoning_effort, "high");
// Pre-#791 strip is preserved for github Claude models that DO NOT opt in
// (Haiku 4.5, Opus 4.7, older Sonnet, etc.).
const githubHaiku = {
model: "claude-haiku-4.5",
reasoning_effort: "max",
messages: [],
};
const githubHaikuResult = sanitizeReasoningEffortForProvider(
githubHaiku,
"github",
"claude-haiku-4.5",
null
);
assert.equal((githubHaikuResult as any).reasoning_effort, undefined);
});
test("sanitizeReasoningEffortForProvider: mistral/devstral strips reasoning object when only effort present", () => {

View File

@@ -0,0 +1,116 @@
// Regression tests for granular reasoning_effort handling on GitHub Copilot
// Claude models (upstream port: decolua/9router#791 by @baslr).
//
// Pre-port behaviour: the github branch of sanitizeReasoningEffortForProvider
// stripped reasoning_effort for ANY model whose name matched /(claude|haiku|oswe)/i,
// so Claude Opus 4.6 and Claude Sonnet 4.6 via GitHub Copilot never received
// extended-thinking configuration even though both backends support it.
//
// Post-port behaviour: reasoning_effort is preserved on Claude Opus 4.6 and
// Claude Sonnet 4.6 (Copilot routes both to Anthropic's chat/completions
// surface where reasoning_effort is honored), and continues to be stripped on
// Claude Haiku 4.5 and Claude Opus 4.7 (rejected upstream).
//
// Note: OmniRoute's openai→claude translator already maps reasoning_effort →
// thinking.budget_tokens far more richly than upstream's tiny effortToBudget
// table (handles `max`, `xhigh`, adaptive models, and fits to max_tokens), so
// only the github-executor half of upstream PR #791 needs porting.
import test from "node:test";
import assert from "node:assert/strict";
const { sanitizeReasoningEffortForProvider } = await import("../../open-sse/executors/base.ts");
test("github/claude-opus-4.6: preserves reasoning_effort (#791)", () => {
const body = {
model: "claude-opus-4.6",
reasoning_effort: "high",
messages: [{ role: "user", content: "hi" }],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-opus-4.6", null);
assert.equal((result as any).reasoning_effort, "high", "Opus 4.6 must keep reasoning_effort");
});
test("github/claude-sonnet-4.6: preserves reasoning_effort (#791)", () => {
const body = {
model: "claude-sonnet-4.6",
reasoning_effort: "medium",
messages: [{ role: "user", content: "hi" }],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-sonnet-4.6", null);
assert.equal((result as any).reasoning_effort, "medium", "Sonnet 4.6 must keep reasoning_effort");
});
test("github/claude-haiku-4.5: still strips reasoning_effort (#791)", () => {
const body = {
model: "claude-haiku-4.5",
reasoning_effort: "high",
messages: [{ role: "user", content: "hi" }],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-haiku-4.5", null);
assert.equal(
(result as any).reasoning_effort,
undefined,
"Haiku 4.5 rejects reasoning_effort upstream — must strip"
);
});
test("github/claude-opus-4.7: still strips reasoning_effort (#791)", () => {
const body = {
model: "claude-opus-4.7",
reasoning_effort: "high",
messages: [{ role: "user", content: "hi" }],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-opus-4.7", null);
assert.equal(
(result as any).reasoning_effort,
undefined,
"Opus 4.7 rejects reasoning_effort upstream — must strip"
);
});
test("github/claude-opus-4.6: preserves nested reasoning.effort (#791)", () => {
const body = {
model: "claude-opus-4.6",
reasoning: { effort: "high", summary: "auto" },
input: [],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-opus-4.6", null);
assert.equal((result as any).reasoning.effort, "high");
assert.equal((result as any).reasoning.summary, "auto", "other reasoning fields preserved");
});
test("github/claude-sonnet-4.5: still strips reasoning_effort (older Sonnet)", () => {
// Upstream PR #791 explicitly opts in only Opus 4.6 and Sonnet 4.6. Older
// Sonnet variants (4.5) keep the historical strip — Copilot has not made
// reasoning_effort available for them.
const body = {
model: "claude-sonnet-4.5",
reasoning_effort: "high",
messages: [{ role: "user", content: "hi" }],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "claude-sonnet-4.5", null);
assert.equal((result as any).reasoning_effort, undefined);
});
test("github/oswe-vscode-prime: still strips reasoning_effort", () => {
// Regression guard: the oswe branch of the rejection pattern must remain.
const body = {
model: "oswe-vscode-prime",
reasoning_effort: "high",
messages: [],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "oswe-vscode-prime", null);
assert.equal((result as any).reasoning_effort, undefined);
});
test("github/gpt-5.4: pass-through (non-Claude unchanged)", () => {
// Regression guard: non-Claude github models keep reasoning_effort.
const body = {
model: "gpt-5.4",
reasoning_effort: "high",
messages: [],
};
const result = sanitizeReasoningEffortForProvider(body, "github", "gpt-5.4", null);
assert.equal((result as any).reasoning_effort, "high");
});