mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-31 12:22:14 +03:00
* chore(release): open v3.8.21 development cycle
* fix: pass through valid max_tokens-truncated responses instead of fake 502 (#3572) (#3595)
* fix: /v1/completions returns legacy text-completion format, not chat (#3571) (#3596)
* fix: z.ai/GLM coding plan no longer shows Monthly 0% when no monthly cap (#3580) (#3597)
* docs: mark DISCOVERY_TOOL_DESIGN endpoints as Phase-2 not-yet-implemented (#3498) (#3599)
* fix(agent-bridge): add validate-only upstream-ca/test route (#3488) (#3600)
* fix(gamification): add level/badges/badges-earned profile routes (#3484)
* security(oauth): migrate 5 public client_ids to resolvePublicCred (#3493)
* fix(mcp): ship MCP server source closure in npm files + coverage gate (#3578)
* fix: add reasoning token buffer for combo routing (fixes #3587) (#3588)
Integrated into release/v3.8.21
* Refactor: Extract chatCore phases into modular files (#3598)
Integrated into release/v3.8.21 — chatCore phase modularization. Adjusted: re-derive idempotencyKey for the save path after the check moved into the module (co-authored). Thanks @oyi77!
* docs(changelog): credit #3598 (chatCore modularization) + #3588 (combo reasoning buffer)
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
* fix(api): implement GET /api/guardrails + POST /api/guardrails/test, drop shadow/guardrails doc-fiction (#3496) (#3602)
Integrated into release/v3.8.21 — implements GET /api/guardrails + POST /api/guardrails/test, removes shadow/guardrails doc-fiction. TDD-validated (5/5) + check-docs-symbols/typecheck/eslint green.
* fix(gemini): isolate textual reasoning wrappers (#3605)
Split-out PR C from #3584. Isolates textual reasoning wrappers (<think>/<thinking>/<thought>/<internal_thought>, including malformed/open tags) into reasoning_content across both the non-streaming sanitizer and the Gemini streaming translator, with split-chunk buffering. Additive to the existing textual tool-call pipeline; does not touch the #3569 native functionResponse path. Integrated into release/v3.8.21. Thanks @dhaern!
* fix(antigravity): normalize Gemini 3.5 Flash tier IDs (#3603)
Split-out PR A from #3584. Normalizes the Antigravity/agy Gemini 3.5 Flash tier IDs to clean public names (gemini-3.5-flash-low/medium/high), maps them to the live upstream IDs at the executor boundary, and removes Antigravity from the global model resolver so the executor owns wire normalization. Maintainer follow-up: kept gemini-3.5-flash-preview as a hidden backward-compat alias routing to the High tier (so saved combos/configs keep working). Live-validated the tier set via the agy CLI catalog. Integrated into release/v3.8.21. Thanks @dhaern!
* fix(agent-bridge): surface real MITM startup-failure cause, not always port 443 (#3606) (#3608)
Integrated into release/v3.8.21 (#3606)
* fix(oauth): surface real Kiro import-token failure cause, not a bare 500 (#3589) (#3609)
Integrated into release/v3.8.21 (#3589)
* docs(opencode-provider): soft-deprecate in favor of @omniroute/opencode-plugin (#3419) (#3613)
Integrated into release/v3.8.21 (#3419)
* fix(usage): normalize Antigravity and agy provider quotas (#3604)
Split-out PR B from #3584. Normalizes Antigravity/agy provider quotas: prefers retrieveUserQuota for live consumption, falls back to fetchAvailableModels and local usage_history, sanitizes cached Provider Limits so retired upstream IDs are not re-exposed, and schedules a deduplicated post-usage refresh. Maintainer follow-up: decoupled the post-usage refresh via a lightweight usageEvents bus (usageHistory no longer dynamic-imports providerLimits) so it does not pull the executors/translator graph into the typecheck-core surface — typecheck:core stays at 0. Integrated into release/v3.8.21. Thanks @dhaern!
* feat(cli): add autostart on/off/toggle shorthand for headless serve mode (#3331) (#3614)
Integrated into release/v3.8.21 (#3331)
* docs(changelog): credit #3603 (Flash tier IDs) + #3604 (provider quotas) + #3605 (reasoning wrappers)
Co-authored-by: diegosouzapw <diegosouza.pw@gmail.com>
* fix(review): resolve findings from /review-reviews battery (v3.8.21 hardening) (#3618)
Pre-release hardening from the /review-reviews battery — 15 findings resolved (L1-L13,L15) + L14 live-verified WONTFIX, convergence re-review clean. lint/typecheck:core/test:vitest(146)/build green; zero new test:unit failures vs baseline 797de433f.
* chore(release): v3.8.21 CHANGELOG + i18n + env-doc sync
---------
Co-authored-by: Hernan Javier Ardila Sanchez <hjasgr@gmail.com>
Co-authored-by: Paijo <14921983+oyi77@users.noreply.github.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
Co-authored-by: Raxxoor <manker_lol@hotmail.com>
177 lines
6.1 KiB
TypeScript
177 lines
6.1 KiB
TypeScript
import {
|
|
isAccountDeactivated,
|
|
isCreditsExhausted,
|
|
isDailyQuotaExhausted,
|
|
isOAuthInvalidToken,
|
|
} from "./accountFallback.ts";
|
|
import { getProviderCategory } from "../config/providerRegistry.ts";
|
|
|
|
// Terminal stop signals where an empty content payload is still a legitimate,
|
|
// successful completion (truncated at the token limit, or a tool-call turn) —
|
|
// NOT a silent "fake success" failure. Used to avoid rewriting a valid HTTP 200
|
|
// (e.g. a Claude Code `max_tokens: 1` connectivity ping) into a synthetic 502.
|
|
const LEGIT_EMPTY_CLAUDE_STOP = new Set(["max_tokens", "tool_use"]);
|
|
const LEGIT_EMPTY_OPENAI_FINISH = new Set(["length", "tool_calls"]);
|
|
|
|
export function isEmptyContentResponse(responseBody: unknown): boolean {
|
|
if (!responseBody || typeof responseBody !== "object") return false;
|
|
|
|
const body = responseBody as Record<string, unknown>;
|
|
|
|
if (Array.isArray(body.choices)) {
|
|
const firstChoice = body.choices[0] as Record<string, unknown> | undefined;
|
|
if (!firstChoice) return true;
|
|
|
|
const message = firstChoice.message as Record<string, unknown> | undefined;
|
|
const delta = firstChoice.delta as Record<string, unknown> | undefined;
|
|
|
|
const content = message?.content ?? delta?.content;
|
|
const reasoningContent = message?.reasoning_content ?? delta?.reasoning_content;
|
|
const hasToolCalls =
|
|
(Array.isArray(message?.tool_calls) && (message.tool_calls as unknown[]).length > 0) ||
|
|
(Array.isArray(delta?.tool_calls) && (delta.tool_calls as unknown[]).length > 0);
|
|
|
|
const hasContent = content !== null && content !== undefined && content !== "";
|
|
const hasReasoning =
|
|
reasoningContent !== null && reasoningContent !== undefined && reasoningContent !== "";
|
|
|
|
// A response truncated at the token limit (finish_reason "length") is a valid,
|
|
// successful completion even with empty text — do not flag it as a fake success.
|
|
const finishReason =
|
|
typeof firstChoice.finish_reason === "string" ? firstChoice.finish_reason : "";
|
|
if (LEGIT_EMPTY_OPENAI_FINISH.has(finishReason)) return false;
|
|
|
|
return !hasContent && !hasReasoning && !hasToolCalls;
|
|
}
|
|
|
|
if (Array.isArray(body.content)) {
|
|
if (body.content.length > 0) return false;
|
|
// Empty content array: a response truncated at max_tokens (or one that stopped
|
|
// to emit a tool_use block) is a legitimate terminal state, not a silent
|
|
// failure. Only flag empty content when no such terminal stop_reason is present.
|
|
const stopReason = typeof body.stop_reason === "string" ? body.stop_reason : "";
|
|
return !LEGIT_EMPTY_CLAUDE_STOP.has(stopReason);
|
|
}
|
|
|
|
if (typeof body.text === "string") {
|
|
return body.text.trim() === "";
|
|
}
|
|
|
|
if ("content" in body) {
|
|
const content = body.content;
|
|
return content === null || content === undefined || content === "";
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
export const PROVIDER_ERROR_TYPES = {
|
|
RATE_LIMITED: "rate_limited",
|
|
UNAUTHORIZED: "unauthorized",
|
|
ACCOUNT_DEACTIVATED: "account_deactivated",
|
|
FORBIDDEN: "forbidden",
|
|
SERVER_ERROR: "server_error",
|
|
QUOTA_EXHAUSTED: "quota_exhausted",
|
|
PROJECT_ROUTE_ERROR: "project_route_error",
|
|
CONTEXT_OVERFLOW: "context_overflow",
|
|
OAUTH_INVALID_TOKEN: "oauth_invalid_token",
|
|
EMPTY_CONTENT: "empty_content",
|
|
};
|
|
|
|
export const CONTEXT_OVERFLOW_SIGNALS = [
|
|
"context overflow",
|
|
"prompt too large",
|
|
"context window",
|
|
"maximum context",
|
|
"exceeds context",
|
|
"input too long",
|
|
"token limit",
|
|
"too many tokens",
|
|
"context length",
|
|
"exceed.*context",
|
|
"messages exceed",
|
|
];
|
|
|
|
export const CONTEXT_OVERFLOW_REGEX = new RegExp(CONTEXT_OVERFLOW_SIGNALS.join("|"), "i");
|
|
|
|
export function isContextOverflow(errorText: string): boolean {
|
|
return CONTEXT_OVERFLOW_REGEX.test(String(errorText || ""));
|
|
}
|
|
|
|
function responseBodyToString(responseBody: unknown): string {
|
|
if (typeof responseBody === "string") return responseBody;
|
|
if (responseBody !== null && typeof responseBody === "object") {
|
|
try {
|
|
return JSON.stringify(responseBody);
|
|
} catch {
|
|
return "";
|
|
}
|
|
}
|
|
return "";
|
|
}
|
|
|
|
function shouldPreserveQuotaSignalsFor429(provider?: string | null): boolean {
|
|
if (!provider) return true;
|
|
return getProviderCategory(provider) === "oauth";
|
|
}
|
|
|
|
export function classifyProviderError(
|
|
statusCode: number,
|
|
responseBody: unknown,
|
|
provider?: string | null
|
|
): string | null {
|
|
const bodyStr = responseBodyToString(responseBody);
|
|
const creditsExhausted = isCreditsExhausted(bodyStr);
|
|
const accountDeactivated = isAccountDeactivated(bodyStr);
|
|
const oauthInvalid = isOAuthInvalidToken(bodyStr);
|
|
const preserveQuota429 = shouldPreserveQuotaSignalsFor429(provider);
|
|
|
|
if (creditsExhausted && [400, 402, 403].includes(statusCode)) {
|
|
return PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED;
|
|
}
|
|
|
|
if (creditsExhausted && statusCode === 429 && preserveQuota429) {
|
|
return PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED;
|
|
}
|
|
|
|
// API-key providers route 429 cooldowns through the resilience-aware fallback layer.
|
|
// OAuth providers keep their existing quota semantics because some of them encode
|
|
// longer quota windows as 429 responses.
|
|
if (statusCode === 429) {
|
|
if (preserveQuota429 && isDailyQuotaExhausted(bodyStr)) {
|
|
return PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED;
|
|
}
|
|
return PROVIDER_ERROR_TYPES.RATE_LIMITED;
|
|
}
|
|
|
|
if (statusCode === 401) {
|
|
if (oauthInvalid) {
|
|
return PROVIDER_ERROR_TYPES.OAUTH_INVALID_TOKEN;
|
|
}
|
|
return accountDeactivated
|
|
? PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED
|
|
: PROVIDER_ERROR_TYPES.UNAUTHORIZED;
|
|
}
|
|
|
|
if (statusCode === 402) return PROVIDER_ERROR_TYPES.QUOTA_EXHAUSTED;
|
|
if (statusCode === 403 && accountDeactivated) {
|
|
return PROVIDER_ERROR_TYPES.ACCOUNT_DEACTIVATED;
|
|
}
|
|
if (statusCode === 403) {
|
|
if (bodyStr.includes("has not been used in project")) {
|
|
return PROVIDER_ERROR_TYPES.PROJECT_ROUTE_ERROR;
|
|
}
|
|
if (provider && getProviderCategory(provider) === "apikey") {
|
|
return null;
|
|
}
|
|
return PROVIDER_ERROR_TYPES.FORBIDDEN;
|
|
}
|
|
if (statusCode >= 500) return PROVIDER_ERROR_TYPES.SERVER_ERROR;
|
|
|
|
if (statusCode === 400 && isContextOverflow(bodyStr)) {
|
|
return PROVIDER_ERROR_TYPES.CONTEXT_OVERFLOW;
|
|
}
|
|
|
|
return null;
|
|
}
|