mirror of
https://github.com/diegosouzapw/OmniRoute.git
synced 2026-07-26 09:52:11 +03:00
* chore(release): open v3.8.18 development cycle * fix(catalog): stop Codex CLI model-catalog refresh from erroring (#3481) Codex's model-catalog refresh (codex_models_manager) does GET /v1/models?client_version=<v> and decodes a JSON object with a TOP-LEVEL `models` array. OmniRoute answers in the OpenAI-standard `{object,data}` shape, so codex fails with "missing field `models`" and logs "failed to refresh available models" on every startup. Detect codex clients via the `originator` / `user-agent` = `codex_*` headers they send and add an EMPTY top-level `models: []` so the decode succeeds. Non-codex OpenAI clients keep the byte-identical `{object,data}` response. The array is intentionally empty: codex replaces its built-in per-model agent prompt (`base_instructions`, ~21k chars) with whatever a populated entry carries for the selected model, so emitting our catalog would drop the agent prompt to nothing and break codex's agent behaviour (verified empirically against codex 0.137). An empty list keeps codex on its built-in model info — same inference as before, minus the error. Validated end-to-end with the real handler against codex 0.137: "failed to refresh available models" → 0 occurrences, instructions preserved (built-in Codex agent prompt, not empty). Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * chore: ignore quality reports and local prompt artifacts Add generated quality gate reports, metrics files, and local setup prompt artifacts to .gitignore to prevent committing environment-specific or temporary files. * fix(provider): detect Responses API format when body has `input` but … (#3490) Integrated into release/v3.8.18 * fix(sse): normalize numeric provider ids to strings (#3451) Integrated into release/v3.8.18 * feat(browserPool): resolve Playwright proxy from proxy_registry DB (#3492) Integrated into release/v3.8.18 * fix(theoldllm): generate X-Request-Token server-side, drop Playwright (#3491) Integrated into release/v3.8.18 * feat(plugins): add lifecycle hooks and theme-manager plugin (#3473) Integrated into release/v3.8.18 * fix(combo): parallel pre-screen + circuit-breaker fast-exit for priority combos (#3169) Integrated into release/v3.8.18 * feat(ui): unifi active and finished requests into single view #1422 (#3401) Integrated into release/v3.8.18 * docs(changelog): record #3401, #3473, #3492, #3490, #3451, #3491, #3169 under v3.8.18 * feat(docs): add doc accuracy gate + refresh AGENTS.md counts (#3510) Integrated into release/v3.8.18 * fix(sse): drop empty-choices chunks without usage instead of injecting retry text (#3513) PR #3422 ('allow OpenAI usage-only empty choices chunks') reintroduced the assistant-content injection '[OmniRoute] Upstream returned an empty response. Please retry.' for empty `choices: []` chunks that carry no valid usage. Clients (Goose/opencode) feed that text back as a turn and spin in a retry loop -- the exact regression #3400 had fixed by dropping the chunk. Restore the drop behavior for the no-usage case while preserving #3422's standards-compliant forwarding of usage-only `include_usage` final chunks. Realign the mislabeled stream-utils test (it asserted the injection) and add a dedicated regression guard. Reported-by: @mochizzan Refs: #3502, #3388, #3400, #3422 * fix(authz): fall back to URL token when Authorization isn't a usable Bearer (#3504) Integrated into release/v3.8.18 * fix(playground): authenticate via session, test key policy by id (#3503) Integrated into release/v3.8.18 * docs(changelog): record #3510, #3504, #3503 under v3.8.18 * fix: llama base url normalization (#3519) * docs(changelog): reconcile v3.8.18 — add #3519, #3513, #3435-repair, gitignore chore (full commit↔changelog coverage) * fix(opencode-plugin): bound regex quantifiers in normaliseFreeLabel (polynomial-ReDoS) CodeQL js/polynomial-redos: unbounded \s* before an anchored \s*$ allowed O(n²) backtracking on attacker-influenced display names. Bounded to {0,8}/{1,8} (ample for any real label spacing). Plugin builds + 254 tests green. * fix(types): restore clean typecheck:core for v3.8.18 release gate - getPendingRequests() typed to real shape (was widened to object) → fixes unknown 'count' in the unified-requests view (#3401) - streamChunks log payload cast to its declared type (callLogs.ts) - preScreenTargets aligned to canonical IsModelAvailable signature (#3169), Promise.resolve-normalized so .catch never hits a bare boolean All 5 gates green: lint(0 err) + typecheck:core + cycles + docs-all + unit + vitest(146). --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com> Co-authored-by: Andrey Borodulin <borodulin@gmail.com> Co-authored-by: Dmitrii Safronov <zimniy@cyberbrain.cc> Co-authored-by: Paijo <14921983+oyi77@users.noreply.github.com> Co-authored-by: PizzaV <103120356+pizzav-xyz@users.noreply.github.com> Co-authored-by: Markus Hartung <mail@hartmark.se> Co-authored-by: Felipe Almeman <4226997+zhiru@users.noreply.github.com>
296 lines
12 KiB
TypeScript
296 lines
12 KiB
TypeScript
import test from "node:test";
|
|
import assert from "node:assert/strict";
|
|
|
|
const {
|
|
checkFallbackError,
|
|
parseRetryAfterFromBody,
|
|
classifyError,
|
|
classifyErrorText,
|
|
lockModel,
|
|
isModelLocked,
|
|
getModelLockoutInfo,
|
|
getAllModelLockouts,
|
|
getQuotaCooldown,
|
|
getBackoffDuration,
|
|
getAccountHealth,
|
|
isAccountUnavailable,
|
|
getUnavailableUntil,
|
|
formatRetryAfter,
|
|
filterAvailableAccounts,
|
|
resetAccountState,
|
|
applyErrorState,
|
|
} = await import("../../open-sse/services/accountFallback.ts");
|
|
|
|
const { RateLimitReason, BACKOFF_STEPS_MS } = await import("../../open-sse/config/constants.ts");
|
|
|
|
// ─── parseRetryAfterFromBody Tests ──────────────────────────────────────────
|
|
|
|
test("parseRetryAfterFromBody: parses Gemini retryDelay format", () => {
|
|
const body = {
|
|
error: {
|
|
code: 429,
|
|
message: "Resource has been exhausted",
|
|
details: [{ "@type": "google.rpc.RetryInfo", retryDelay: "33s" }],
|
|
},
|
|
};
|
|
const result = parseRetryAfterFromBody(body);
|
|
assert.equal(result.retryAfterMs, 33000);
|
|
assert.equal(result.reason, RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
});
|
|
|
|
test("parseRetryAfterFromBody: parses OpenAI retry message format", () => {
|
|
const body = {
|
|
error: {
|
|
message: "Rate limit reached. Please retry after 20s.",
|
|
type: "rate_limit_error",
|
|
},
|
|
};
|
|
const result = parseRetryAfterFromBody(body);
|
|
assert.equal(result.retryAfterMs, 20000);
|
|
assert.equal(result.reason, RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
});
|
|
|
|
test("parseRetryAfterFromBody: classifies Anthropic rate_limit_error", () => {
|
|
const body = {
|
|
type: "error",
|
|
error: { type: "rate_limit_error", message: "Too many requests" },
|
|
};
|
|
const result = parseRetryAfterFromBody(body);
|
|
assert.equal(result.reason, RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
});
|
|
|
|
test("parseRetryAfterFromBody: handles string input", () => {
|
|
const body = JSON.stringify({
|
|
error: { details: [{ retryDelay: "10s" }] },
|
|
});
|
|
const result = parseRetryAfterFromBody(body);
|
|
assert.equal(result.retryAfterMs, 10000);
|
|
});
|
|
|
|
test("parseRetryAfterFromBody: handles invalid JSON", () => {
|
|
const result = parseRetryAfterFromBody("not json");
|
|
assert.equal(result.retryAfterMs, null);
|
|
assert.equal(result.reason, RateLimitReason.UNKNOWN);
|
|
});
|
|
|
|
test("parseRetryAfterFromBody: handles null/undefined", () => {
|
|
assert.equal(parseRetryAfterFromBody(null).retryAfterMs, null);
|
|
assert.equal(parseRetryAfterFromBody(undefined).retryAfterMs, null);
|
|
});
|
|
|
|
// ─── classifyError Tests ────────────────────────────────────────────────────
|
|
|
|
test("classifyError: 429 → RATE_LIMIT_EXCEEDED", () => {
|
|
assert.equal(classifyError(429, ""), RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
});
|
|
|
|
test("classifyError: 401 → AUTH_ERROR", () => {
|
|
assert.equal(classifyError(401, ""), RateLimitReason.AUTH_ERROR);
|
|
});
|
|
|
|
test("classifyError: 402 → QUOTA_EXHAUSTED", () => {
|
|
assert.equal(classifyError(402, ""), RateLimitReason.QUOTA_EXHAUSTED);
|
|
});
|
|
|
|
test("classifyError: 503 → MODEL_CAPACITY", () => {
|
|
assert.equal(classifyError(503, ""), RateLimitReason.MODEL_CAPACITY);
|
|
});
|
|
|
|
test("classifyError: text overrides status code", () => {
|
|
// 500 normally → SERVER_ERROR, but quota text → QUOTA_EXHAUSTED
|
|
assert.equal(classifyError(500, "quota exceeded"), RateLimitReason.QUOTA_EXHAUSTED);
|
|
});
|
|
|
|
test("classifyErrorText: handles various patterns", () => {
|
|
assert.equal(classifyErrorText("rate limit reached"), RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
assert.equal(classifyErrorText("too many requests"), RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
assert.equal(classifyErrorText("capacity exceeded"), RateLimitReason.MODEL_CAPACITY);
|
|
assert.equal(classifyErrorText("overloaded"), RateLimitReason.MODEL_CAPACITY);
|
|
assert.equal(classifyErrorText("unauthorized"), RateLimitReason.AUTH_ERROR);
|
|
assert.equal(classifyErrorText("random error"), RateLimitReason.UNKNOWN);
|
|
});
|
|
|
|
test("classifyErrorText: Gemini 503 high demand maps to MODEL_CAPACITY", () => {
|
|
const geminiMsg =
|
|
"[503]: This model is currently experiencing high demand. Spikes in demand are usually temporary. Please try again later.";
|
|
assert.equal(classifyErrorText(geminiMsg), RateLimitReason.MODEL_CAPACITY);
|
|
});
|
|
|
|
test("classifyError: 503 with Gemini high demand message returns MODEL_CAPACITY", () => {
|
|
const geminiMsg =
|
|
"[503]: This model is currently experiencing high demand. Spikes in demand are usually temporary. Please try again later.";
|
|
assert.equal(classifyError(503, geminiMsg), RateLimitReason.MODEL_CAPACITY);
|
|
});
|
|
|
|
test("checkFallbackError: 503 with Gemini high demand returns MODEL_CAPACITY reason", () => {
|
|
const geminiMsg =
|
|
"[503]: This model is currently experiencing high demand. Spikes in demand are usually temporary. Please try again later.";
|
|
const result = checkFallbackError(503, geminiMsg);
|
|
assert.equal(result.shouldFallback, true);
|
|
assert.equal(result.reason, RateLimitReason.MODEL_CAPACITY);
|
|
assert.ok(result.cooldownMs > 0, "cooldownMs should be positive");
|
|
});
|
|
|
|
// ─── Per-Model Lockout Tests ────────────────────────────────────────────────
|
|
|
|
test("lockModel + isModelLocked: locks specific model", () => {
|
|
lockModel("claude", "conn1", "claude-sonnet-4", "rate_limit_exceeded", 5000);
|
|
assert.equal(isModelLocked("claude", "conn1", "claude-sonnet-4"), true);
|
|
});
|
|
|
|
test("isModelLocked: different model not locked", () => {
|
|
lockModel("claude", "conn2", "claude-sonnet-4", "rate_limit_exceeded", 5000);
|
|
assert.equal(isModelLocked("claude", "conn2", "claude-haiku-4"), false);
|
|
});
|
|
|
|
test("isModelLocked: returns false when no model specified", () => {
|
|
assert.equal(isModelLocked("claude", "conn1", null), false);
|
|
assert.equal(isModelLocked("claude", "conn1", undefined), false);
|
|
});
|
|
|
|
test("getModelLockoutInfo: returns lockout details", () => {
|
|
lockModel("openai", "conn3", "gpt-4o", "quota_exhausted", 10000);
|
|
const info = getModelLockoutInfo("openai", "conn3", "gpt-4o");
|
|
assert.ok(info);
|
|
assert.equal(info.reason, "quota_exhausted");
|
|
assert.ok(info.remainingMs > 0);
|
|
});
|
|
|
|
test("getAllModelLockouts: returns active lockouts", () => {
|
|
lockModel("test-provider", "conn-test", "test-model", "test", 10000);
|
|
const lockouts = getAllModelLockouts();
|
|
const found = lockouts.find((l) => l.model === "test-model");
|
|
assert.ok(found);
|
|
assert.equal(found.provider, "test-provider");
|
|
});
|
|
|
|
// ─── checkFallbackError Tests ────────────────────────────────────────────────
|
|
|
|
test("checkFallbackError: backward compatible without model param", () => {
|
|
const result = checkFallbackError(429, "Rate limit hit", 0);
|
|
assert.equal(result.shouldFallback, true);
|
|
assert.ok(result.cooldownMs > 0);
|
|
assert.equal(result.newBackoffLevel, 1);
|
|
assert.equal(result.reason, RateLimitReason.RATE_LIMIT_EXCEEDED);
|
|
});
|
|
|
|
test("checkFallbackError: 400 does not trigger fallback", () => {
|
|
const result = checkFallbackError(400, "bad request");
|
|
assert.equal(result.shouldFallback, false);
|
|
});
|
|
|
|
test("checkFallbackError: server error has reason", () => {
|
|
const result = checkFallbackError(500, "internal server error");
|
|
assert.equal(result.shouldFallback, true);
|
|
assert.equal(result.reason, RateLimitReason.SERVER_ERROR);
|
|
});
|
|
|
|
test("checkFallbackError: transient errors now apply exponential backoff", () => {
|
|
const result = checkFallbackError(502, "", 5);
|
|
assert.equal(result.shouldFallback, true);
|
|
assert.equal(result.newBackoffLevel, 6); // Backoff now increments for transients
|
|
assert.ok(result.cooldownMs > 0, "cooldownMs should be positive");
|
|
});
|
|
|
|
// ─── Backoff Steps Tests ────────────────────────────────────────────────────
|
|
|
|
test("getBackoffDuration: follows step sequence", () => {
|
|
assert.equal(getBackoffDuration(0), BACKOFF_STEPS_MS[0]); // 60s
|
|
assert.equal(getBackoffDuration(1), BACKOFF_STEPS_MS[1]); // 120s
|
|
assert.equal(getBackoffDuration(2), BACKOFF_STEPS_MS[2]); // 300s
|
|
assert.equal(getBackoffDuration(4), BACKOFF_STEPS_MS[4]); // 1200s
|
|
});
|
|
|
|
test("getBackoffDuration: caps at max step", () => {
|
|
assert.equal(getBackoffDuration(100), BACKOFF_STEPS_MS[BACKOFF_STEPS_MS.length - 1]);
|
|
});
|
|
|
|
// ─── Exponential backoff (original) Tests ───────────────────────────────────
|
|
|
|
test("getQuotaCooldown: exponential progression", () => {
|
|
assert.equal(getQuotaCooldown(0), 1000); // 1s
|
|
assert.equal(getQuotaCooldown(1), 2000); // 2s
|
|
assert.equal(getQuotaCooldown(3), 8000); // 8s
|
|
assert.ok(getQuotaCooldown(20) <= 120000); // Capped at 2min
|
|
});
|
|
|
|
// ─── Account Health Tests ───────────────────────────────────────────────────
|
|
|
|
test("getAccountHealth: healthy account = 100", () => {
|
|
assert.equal(getAccountHealth({ backoffLevel: 0 }), 100);
|
|
});
|
|
|
|
test("getAccountHealth: degraded by backoff level", () => {
|
|
assert.equal(getAccountHealth({ backoffLevel: 5 }), 50);
|
|
});
|
|
|
|
test("getAccountHealth: degraded by error + rateLimited", () => {
|
|
const score = getAccountHealth({
|
|
backoffLevel: 3,
|
|
lastError: { message: "something" },
|
|
rateLimitedUntil: new Date(Date.now() + 60000).toISOString(),
|
|
});
|
|
assert.equal(score, 20); // 100 - 30 - 20 - 30
|
|
});
|
|
|
|
test("getAccountHealth: null account = 0", () => {
|
|
assert.equal(getAccountHealth(null), 0);
|
|
});
|
|
|
|
// ─── Account State Tests ────────────────────────────────────────────────────
|
|
|
|
test("resetAccountState: clears all error state", () => {
|
|
const reset = resetAccountState({
|
|
rateLimitedUntil: "2030-01-01",
|
|
backoffLevel: 5,
|
|
lastError: "something",
|
|
status: "error",
|
|
});
|
|
assert.equal(reset.rateLimitedUntil, null);
|
|
assert.equal(reset.backoffLevel, 0);
|
|
assert.equal(reset.lastError, null);
|
|
assert.equal(reset.status, "active");
|
|
});
|
|
|
|
test("applyErrorState: applies cooldown and reason", () => {
|
|
const result = applyErrorState({ backoffLevel: 0 }, 429, "rate limit hit");
|
|
assert.ok(result.rateLimitedUntil);
|
|
assert.equal(result.backoffLevel, 1);
|
|
assert.equal(result.status, "error");
|
|
assert.ok(result.lastError.reason);
|
|
});
|
|
|
|
// ─── Utility Tests ──────────────────────────────────────────────────────────
|
|
|
|
test("isAccountUnavailable: false for null", () => {
|
|
assert.equal(isAccountUnavailable(null), false);
|
|
});
|
|
|
|
test("isAccountUnavailable: true for future timestamp", () => {
|
|
assert.equal(isAccountUnavailable(new Date(Date.now() + 60000).toISOString()), true);
|
|
});
|
|
|
|
test("isAccountUnavailable: false for past timestamp", () => {
|
|
assert.equal(isAccountUnavailable(new Date(Date.now() - 1000).toISOString()), false);
|
|
});
|
|
|
|
test("formatRetryAfter: formats correctly", () => {
|
|
const future = new Date(Date.now() + 150000).toISOString(); // 2.5 min
|
|
const formatted = formatRetryAfter(future);
|
|
assert.match(formatted, /reset after \d+m/);
|
|
});
|
|
|
|
test("filterAvailableAccounts: filters out rate-limited", () => {
|
|
const accounts = [
|
|
{ id: "a", rateLimitedUntil: null },
|
|
{ id: "b", rateLimitedUntil: new Date(Date.now() + 60000).toISOString() },
|
|
{ id: "c", rateLimitedUntil: new Date(Date.now() - 1000).toISOString() },
|
|
];
|
|
const available = filterAvailableAccounts(accounts);
|
|
assert.equal(available.length, 2); // a and c (expired)
|
|
assert.deepEqual(
|
|
available.map((a) => a.id),
|
|
["a", "c"]
|
|
);
|
|
});
|