Compare commits
97 Commits
fix/codeql
...
release/v3
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f95b03d709 | ||
|
|
1d0c5a36db | ||
|
|
b24cc53d54 | ||
|
|
59dccdd9e1 | ||
|
|
9464792cfc | ||
|
|
11cbd7d4e0 | ||
|
|
f93fecd86b | ||
|
|
d5d730c845 | ||
|
|
8bbe92c692 | ||
|
|
bbc7bf4351 | ||
|
|
56d64e29a4 | ||
|
|
7b36e45df8 | ||
|
|
ddee064f1b | ||
|
|
b1fdfd5ea4 | ||
|
|
71eeaf293c | ||
|
|
406f4524ff | ||
|
|
dfc5b5eec4 | ||
|
|
0a53c8a2ce | ||
|
|
93da24cd79 | ||
|
|
815c7c2864 | ||
|
|
8f15b79a84 | ||
|
|
07a378c86c | ||
|
|
34150506f2 | ||
|
|
76ac1c8b7e | ||
|
|
d732cf615d | ||
|
|
f58e8bef6f | ||
|
|
243445f210 | ||
|
|
13e29f2f39 | ||
|
|
2544ee9498 | ||
|
|
440113c8e8 | ||
|
|
3c2906a80e | ||
|
|
095f424658 | ||
|
|
20de0d9c79 | ||
|
|
9f30b76057 | ||
|
|
019ad33a61 | ||
|
|
dfc9257b07 | ||
|
|
37e71915db | ||
|
|
077bc1a8a2 | ||
|
|
378eff0f75 | ||
|
|
6de542b9b6 | ||
|
|
315b0a94e1 | ||
|
|
e1c2b347f9 | ||
|
|
f88aa48847 | ||
|
|
8301984734 | ||
|
|
028f1b91e4 | ||
|
|
2af1326adf | ||
|
|
644dd32d3f | ||
|
|
9df3f8923d | ||
|
|
0b7ac870ef | ||
|
|
9fedc1c411 | ||
|
|
e589831952 | ||
|
|
04d2a60331 | ||
|
|
d23bfefec0 | ||
|
|
c8ad44e018 | ||
|
|
c83116e634 | ||
|
|
7715825cb8 | ||
|
|
761d38f433 | ||
|
|
c6963ca5dd | ||
|
|
b010d8bf86 | ||
|
|
fdcd15e6a9 | ||
|
|
12b8df02dd | ||
|
|
38d21afc2d | ||
|
|
05e76d6e76 | ||
|
|
93135f8e18 | ||
|
|
22086a73fa | ||
|
|
d4ade9d1d3 | ||
|
|
f54c93c879 | ||
|
|
e2e48fdab8 | ||
|
|
d2cea0811a | ||
|
|
2f18a85310 | ||
|
|
38969ad16b | ||
|
|
dafb4ae808 | ||
|
|
338c05dc6a | ||
|
|
6945bbaaba | ||
|
|
690f684bfc | ||
|
|
c3cd1f94c0 | ||
|
|
c21460f22a | ||
|
|
9b14896a6c | ||
|
|
29f26293c3 | ||
|
|
cb11592441 | ||
|
|
5ee646e68e | ||
|
|
6984676d95 | ||
|
|
79f8ae9d1e | ||
|
|
04b2c47940 | ||
|
|
24ac71465e | ||
|
|
8d6f91b558 | ||
|
|
c3698eedcb | ||
|
|
adca3b881c | ||
|
|
ac02c5b42f | ||
|
|
07d1816a45 | ||
|
|
65e81158ab | ||
|
|
c68cda7dfb | ||
|
|
ca23eed77c | ||
|
|
5f0a394091 | ||
|
|
918fba5e39 | ||
|
|
026e1cadaa | ||
|
|
b090b601a5 |
19
.env.example
@@ -1027,6 +1027,16 @@ PROVIDER_LIMITS_SYNC_SPACING_MS=1500
|
||||
# Used by: open-sse/services/compression/engines/rtk/filterLoader.ts. Default: 0.
|
||||
#OMNIROUTE_RTK_TRUST_PROJECT_FILTERS=0
|
||||
|
||||
# Maximum concurrent synchronous compression workers. Excess jobs wait FIFO.
|
||||
# Used by: open-sse/services/compression/compressionWorkerPool.ts. Default: 2.
|
||||
#OMNI_COMPRESSION_WORKERS=2
|
||||
# Per-job worker timeout (ms). A timed-out worker is terminated and the request fails open.
|
||||
# Used by: open-sse/services/compression/compressionWorkerPool.ts. Default: 120000.
|
||||
#OMNI_COMPRESSION_WORKER_TIMEOUT_MS=120000
|
||||
# Terminate idle compression workers after this many milliseconds.
|
||||
# Used by: open-sse/services/compression/compressionWorkerPool.ts. Default: 60000.
|
||||
#OMNI_COMPRESSION_WORKER_IDLE_MS=60000
|
||||
|
||||
# T02 stacked-pipeline engine circuit-breaker (OPT-IN, default off). When enabled, a compression
|
||||
# engine that throws repeatedly across requests is skipped (fail-open) for a cooldown.
|
||||
# Used by: open-sse/services/compression/pipelineEngineBreaker.ts.
|
||||
@@ -2011,6 +2021,9 @@ APP_LOG_TO_FILE=true
|
||||
# CLIPROXYAPI_HOST=127.0.0.1
|
||||
# CLIPROXYAPI_PORT=5544
|
||||
# CLIPROXYAPI_CONFIG_DIR=~/.cli-proxy-api
|
||||
# Management key for an externally managed instance. Embedded instances use
|
||||
# OmniRoute's encrypted service key.
|
||||
# CLIPROXYAPI_MANAGEMENT_KEY=
|
||||
|
||||
# ── Mux embedded service ──
|
||||
# Override the port where the embedded Mux (coder/mux) agent-orchestration
|
||||
@@ -2420,10 +2433,10 @@ APP_LOG_TO_FILE=true
|
||||
# test suite must NEVER mutate the OS trust store (a fake test PEM installed via
|
||||
# update-ca-certificates broke all system TLS on a persistent runner, 2026-07-05).
|
||||
# OMNIROUTE_SKIP_SYSTEM_TRUST=1
|
||||
# check-changelog-integrity.mjs (anti CHANGELOG-eat gate): explicit base ref
|
||||
# override, and the justified-removal escape hatch for intentional bullet removals.
|
||||
# check-changelog-integrity.mjs (anti CHANGELOG-eat gate): explicit base ref override.
|
||||
# Intentional transformations require an exact reviewed entry in
|
||||
# config/release/changelog-reconciliations.json; there is no runtime bypass.
|
||||
# CHANGELOG_BASE_REF=origin/release/v0.0.0
|
||||
# ALLOW_CHANGELOG_REMOVALS=1
|
||||
|
||||
# ── Remote audio provider nodes ──
|
||||
# Used by: src/app/api/v1/_shared/audioProviderNodes.ts — lets the /v1/audio/*
|
||||
|
||||
1
.gitignore
vendored
@@ -14,6 +14,7 @@ _tasks/
|
||||
.agents/**
|
||||
.claude/**
|
||||
.gemini/**
|
||||
.code-forge/**
|
||||
.config/**
|
||||
.data/**
|
||||
.logs/**
|
||||
|
||||
@@ -180,6 +180,7 @@ _Living section — regenerated 2026-08-12 from all cycle commits (cycle open `e
|
||||
|
||||
### 🐛 Bug Fixes
|
||||
|
||||
- **fix(build):** every route no longer answers HTTP 500 on artifacts built from the release tip ([#11343](https://github.com/diegosouzapw/OmniRoute/issues/11343)) — `next.config.mjs` aliased `better-sqlite3` to its build-time stub **unconditionally**, on the premise that `serverExternalPackages` still won at runtime. It does not: a Turbopack `resolveAlias` rewrites the request *before* the externals check, so the request stopped matching the `better-sqlite3` external entry and the stub was baked into the shipped bundle. The sync driver then failed with `r(...) is not a constructor`, fell through `node:sqlite` and sql.js, and the instrumentation hook aborted at boot. Same failure shape as [#6344](https://github.com/diegosouzapw/OmniRoute/issues/6344), so it gets the same treatment: the alias is opt-in via `OMNIROUTE_BETTER_SQLITE3_STUB=1` through the shared `scripts/build/better-sqlite3-stub-flag.mjs` helper — set it only on a build host that actually hits the SIGABRT build-worker teardown ([#10060](https://github.com/diegosouzapw/OmniRoute/issues/10060)); default builds externalize the real native addon. Regression guards: `tests/unit/better-sqlite3-stub-alias-11343.test.mjs` (5) and the env matrix in `tests/unit/next-config.test.ts`.
|
||||
- **security(search)**: block SSRF via `/v1/search` `provider_options.baseUrl` for the Firecrawl search provider — the client-controlled override is now validated as a public URL before it is used to build the server-side fetch target, so a caller with a valid API key can no longer redirect search requests at loopback, RFC1918, or cloud-metadata hosts — thanks @zmf963
|
||||
- **providers**: honor `PATCH /api/providers/[id]` so `omniroute providers rotate` stops 405ing (the OpenAPI spec and CLI already use PATCH) (PR #10366)
|
||||
- **cli**: route provider test commands through configured connection test endpoints (#10570)
|
||||
|
||||
21
Dockerfile
@@ -181,10 +181,23 @@ ENV NODE_OPTIONS="--max-old-space-size=${OMNIROUTE_BUILD_MEMORY_MB}"
|
||||
# workers for page-data collection (31 on a 32-core builder); on memory-tight
|
||||
# hosts 31 workers + webpack's multi-GB heap blow past RAM and a worker dies
|
||||
# with SIGSEGV at teardown ("worker exited with code: null and signal: SIGSEGV"),
|
||||
# silently leaving no standalone bundle. Next derives the default worker count
|
||||
# from CIRCLE_NODE_TOTAL (workers = N-1), so N=8 → 7 workers: fast enough while
|
||||
# fitting comfortably in RAM on any host. (#10060)
|
||||
ENV CIRCLE_NODE_TOTAL=8
|
||||
# silently leaving no standalone bundle. Next derives the worker count from
|
||||
# CIRCLE_NODE_TOTAL (workers = N-1). (#10060)
|
||||
#
|
||||
# Lowered 8 → 3 (7 workers → 2). Every page-data worker inherits NODE_OPTIONS
|
||||
# above, so the ceiling is per PROCESS, not per build: 7 workers on a 16 GB
|
||||
# GitHub runner (ubuntu-24.04 / ubuntu-24.04-arm, 4 vCPU) exhausted the host and
|
||||
# buildkit failed the whole step with `ResourceExhausted: ... cannot allocate
|
||||
# memory`. The compile phase always finished ("✓ Compiled successfully in
|
||||
# 4.2min"); the kernel killed the build right after "Collecting page data using
|
||||
# 7 workers". It was intermittent for a while and went 100% on 2026-08-22, which
|
||||
# is what a threshold being crossed by ordinary codebase growth looks like.
|
||||
# tests/unit/docker-build-memory-budget.test.ts does the arithmetic and fails if
|
||||
# either knob is raised past what a 16 GB runner holds. 2 workers also stops
|
||||
# oversubscribing the runner's 4 vCPU, which 7 did. Override for a big builder:
|
||||
# `--build-arg OMNIROUTE_BUILD_WORKERS=8`.
|
||||
ARG OMNIROUTE_BUILD_WORKERS=3
|
||||
ENV CIRCLE_NODE_TOTAL=${OMNIROUTE_BUILD_WORKERS}
|
||||
|
||||
COPY . ./
|
||||
RUN --mount=type=cache,id=s/92ca8a61-c1ba-421f-a389-d48ac7258c2d-next-cache,target=/app/.build/next/cache \
|
||||
|
||||
407
README.md
@@ -17,9 +17,9 @@
|
||||
|
||||
</div>
|
||||
|
||||
> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute aggregates the **documented** free tiers of **42 provider pools / 495 models** into one honest number and shows it live on the dashboard (`/dashboard/free-tiers`).
|
||||
> Stacking free tiers by hand is painful — dozens of SDKs, dozens of rate limits, and no idea how much you actually have. OmniRoute catalogs **455 free-tier entries across 40 recurring pool keys** and computes the token headline from the **20 pools with a published positive monthly budget**, deduplicated by shared pool. The result stays visible on the dashboard (`/dashboard/free-tiers`).
|
||||
|
||||
<img src="./docs/diagrams/free-tier-budget.svg" width="100%" alt="OmniRoute free-tier budget card: ~1.51B free tokens per month steady, up to ~2.13B in the first month with signup credits, from the documented free tiers of 42 provider pools / 495 models behind one endpoint. Honest pool-deduped math — each shared pool counted once (counting every rate limit 24/7 would read ~10B; not published), 15 providers ToS-flagged so you decide. Budget bar of the countable free pools with per-model grid (Mistral Large 3 1B, GPT-4o mini 150M, Gemini 2.5 Flash 60M … Claude Sonnet 4.5 25K), one-time first-month signup credits (vertex 300M, agentrouter 200M, predibase 25M, together 25M, glm-cn 20M, doubao 15M, ai21 10M, longcat 10M, deepseek 5M, hyperbolic 5M, nscale 5M), plus permanently-free no-token-cap providers (SiliconFlow, Z.AI GLM-Flash, Kilo, OpenCode Zen, baidu …) and a $10 OpenRouter top-up unlocking +24M/mo — surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers."/>
|
||||
<img src="./docs/diagrams/free-tier-budget.svg" width="100%" alt="OmniRoute free-tier budget card: ~1.51B free tokens per month steady, up to ~2.13B in the first month with signup credits, from 40 documented recurring pool keys covering 455 cataloged free-tier entries behind one endpoint. Honest pool-deduped math — each shared pool counted once, including 20 recurring pools with a published positive monthly token budget; 15 providers are marked avoid in the terms-risk catalog so you decide. Budget bar includes Mistral 1B, LLM7 150M, Nara 150M, Gemini 60M and smaller pools, plus first-month signup credits and permanently-free no-token-cap providers surfaced separately so they never inflate the headline. Live used/remaining on /dashboard/free-tiers."/>
|
||||
|
||||
> Animated summary of the live `/dashboard/free-tiers` page. Full methodology (pool dedupe, credit tiers, provider terms): **[docs/reference/FREE_TIERS.md](docs/reference/FREE_TIERS.md)**.
|
||||
>
|
||||
@@ -61,14 +61,14 @@
|
||||
|
||||
<div align="center">
|
||||
|
||||
| | v3.8.49 | **v3.8.50** | `v3.8.51+` |
|
||||
| ------------------------- | :-----: | :---------: | :---------: |
|
||||
| 🌐 Providers | 290 | **342** | more queued |
|
||||
| 🧠 Documented models | 1185 | **1202** | — |
|
||||
| 🖼️ Modality Bridge | — | 🆕 vision | video |
|
||||
| 📡 Radar free catalog | — | 🆕 opt-in | — |
|
||||
| ⚖️ Quota-aware scheduling | — | — | 🔭 next |
|
||||
| 📊 Quota telemetry | — | — | 🔭 next |
|
||||
| | v3.8.49 | **v3.8.50** | `v3.8.51+` |
|
||||
| ------------------------- | :-----: | :-----------------------: | :---------: |
|
||||
| 🌐 Providers | 290 | **350** | more queued |
|
||||
| 🧠 Unique chat model IDs | 1185 | **1312** | — |
|
||||
| 🖼️ Modality Bridge | — | 🆕 vision + audio + video | — |
|
||||
| 📡 Radar free catalog | — | 🆕 opt-in | — |
|
||||
| ⚖️ Quota-aware scheduling | — | 🆕 Quota-Share | — |
|
||||
| 📊 Quota telemetry | — | 🆕 live | — |
|
||||
|
||||
**→ [Roadmap](ROADMAP.md) — riding the rail to `v3.9.0 LTS`**
|
||||
|
||||
@@ -101,7 +101,7 @@
|
||||
<tr>
|
||||
<td align="right"><b>⚙️ Features</b></td>
|
||||
<td align="center"><a href="#-combos--the-flagship">🎯 Combos</a></td>
|
||||
<td align="center"><a href="#-349-ai-providers--90-free">🌐 Providers</a></td>
|
||||
<td align="center"><a href="#-350-ai-providers--154-catalog-marked-free">🌐 Providers</a></td>
|
||||
<td align="center"><a href="#-full-cli--a2a--mcp">🔌 CLI & MCP</a></td>
|
||||
</tr>
|
||||
<tr>
|
||||
@@ -126,7 +126,7 @@
|
||||
<td align="right"><b>📦 Project</b></td>
|
||||
<td align="center"><a href="#%EF%B8%8F-tech-stack">🛠️ Tech Stack</a></td>
|
||||
<td align="center"><a href="#-documentation">📖 Docs</a></td>
|
||||
<td align="center"><a href="#-500-contributors">👥 Contributors</a></td>
|
||||
<td align="center"><a href="#-600-contributors">👥 Contributors</a></td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
@@ -210,7 +210,7 @@ curl http://localhost:20128/v1/chat/completions \
|
||||
|
||||
</div>
|
||||
|
||||
<img src="./docs/diagrams/promise-pillars.svg" width="100%" alt="The Promise — One endpoint. 350 providers. Never stop building — OmniRoute picks the cheapest one that works. Six pillars: Never hit limits (auto-fallback across 350 providers in milliseconds, zero downtime) · Save up to 95% tokens (RTK + Caveman stacked compression cuts 15–95%, ~89% avg on tool-heavy sessions) · $0 to start (90+ free tiers, 56 free forever — no card needed) · Every tool works (33 coding agents through one config) · One endpoint (OpenAI ↔ Claude ↔ Gemini ↔ Responses API at /v1) · Production-grade (circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals — 25,000+ tests)."/>
|
||||
<img src="./docs/diagrams/promise-pillars.svg" width="100%" alt="The Promise — One endpoint and 350 providers. Automatic fallback keeps routing while another healthy target is available. Six pillars: resilient fallback across 350 providers · up to 95% token savings on eligible workloads · $0 to start with 90+ free tiers and 56 recurring/keyless free-forever providers · 35 CLI/agent integrations through one config · OpenAI, Claude, Gemini and Responses API compatibility at /v1 · production controls including circuit breakers, TLS stealth, MCP 110 tools, A2A, memory, guardrails, evals and 39,000+ static test declarations across 5,100+ tracked test files."/>
|
||||
|
||||
<br/>
|
||||
<br/>
|
||||
@@ -225,7 +225,7 @@ curl http://localhost:20128/v1/chat/completions \
|
||||
|
||||
<div align="center">
|
||||
|
||||
<img src="./docs/diagrams/tier-cascade.svg" width="100%" alt="OmniRoute request flow: your IDE or CLI (Claude Code, Cursor, Cline…) calls one local endpoint (http://localhost:20128/v1); the OmniRoute Smart Router (RTK + Caveman compression, 19 routing strategies, circuit breakers, TLS stealth, MCP, A2A, guardrails) auto-falls back across 4 provider tiers — Tier 1 Subscription (Claude Code, Codex, Copilot), quota out? Tier 2 API Key (DeepSeek, Groq, xAI), budget hit? Tier 3 Cheap (GLM $0.5, MiniMax $0.2), budget hit? Tier 4 Free (Kiro, Qoder, Pollinations) — always on."/>
|
||||
<img src="./docs/diagrams/tier-cascade.svg" width="100%" alt="OmniRoute request flow: your IDE or CLI (Claude Code, Cursor, Cline…) calls one local endpoint (http://localhost:20128/v1); the OmniRoute Smart Router (RTK + Caveman compression, 19 routing strategies, circuit breakers, TLS stealth, MCP, A2A, guardrails) can fall back across 4 provider tiers while an eligible healthy target remains — Tier 1 Subscription, Tier 2 API Key, Tier 3 Cheap and Tier 4 Free."/>
|
||||
|
||||
</div>
|
||||
|
||||
@@ -318,7 +318,7 @@ curl http://localhost:20128/v1/chat/completions \
|
||||
|
||||
<img src="./docs/diagrams/strategies-grid.svg" width="100%" alt="All 19 combo routing strategies animated — one tile per strategy: priority, fill-first, weighted, round-robin, p2c, least-used, random, strict-random, cost-optimized, headroom, reset-window, reset-aware, context-relay, context-optimized, cache-optimized, lkgp, auto, fusion, pipeline. See the table above for what each one does."/>
|
||||
|
||||
> A **combo** is a chain of models OmniRoute routes across **automatically**. Quota runs out, a provider fails, or costs spike — the combo silently slides to the next model. **This is what makes OmniRoute unbreakable.** 🛡️
|
||||
> A **combo** is a chain of models OmniRoute routes across **automatically**. If quota runs out, a provider fails, or costs spike, the combo can move to the next eligible healthy model. 🛡️
|
||||
|
||||
### ⚡ Zero-config — just use `auto`
|
||||
|
||||
@@ -429,7 +429,7 @@ All **19** strategies — mix & match per combo step:
|
||||
<tr>
|
||||
<td align="center">17</td>
|
||||
<td nowrap><code>auto</code></td>
|
||||
<td>14-factor live scoring across every connection 🤖</td>
|
||||
<td>15-factor live scoring across every connection 🤖</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center">18</td>
|
||||
@@ -443,7 +443,7 @@ All **19** strategies — mix & match per combo step:
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<sub>The Auto-Combo engine scores every candidate on **14 factors** (health, quota, cost, latency, success rate, freshness…) — see [`docs/routing/AUTO-COMBO.md`](docs/routing/AUTO-COMBO.md).</sub>
|
||||
<sub>The Auto-Combo engine scores every candidate on **15 factors** (health, quota, cost, latency, task fit, quality, session availability…) — see [`docs/routing/AUTO-COMBO.md`](docs/routing/AUTO-COMBO.md).</sub>
|
||||
|
||||
##
|
||||
|
||||
@@ -461,7 +461,7 @@ All **19** strategies — mix & match per combo step:
|
||||
|
||||
</div>
|
||||
|
||||
<img src="./docs/diagrams/comparison-table.svg" width="100%" alt="What sets OmniRoute apart — comparison table vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 350 providers, 90+ free providers built-in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA, 43 i18n UI locales, 100% MIT self-hosted. OmniRoute is the only one with the full set; competitors show a mix of checks, partials and crosses. Verified from each project's docs."/>
|
||||
<img src="./docs/diagrams/comparison-table.svg" width="100%" alt="What sets OmniRoute apart — a dated feature snapshot vs 9router, OpenRouter, CLIProxyAPI and LiteLLM across 13 capabilities. OmniRoute: 350 providers, 90+ free tiers built in, 19 routing strategies, 12-engine token compression, built-in MCP server with 110 tools, A2A agent protocol, persistent memory, guardrails, cloud agents, TLS fingerprint stealth, Desktop/Termux/PWA and 43 i18n UI locales. OmniRoute is MIT-licensed and self-hostable. Competitor capabilities and counts may change; see the linked methodology."/>
|
||||
|
||||
<sub>📊 Full methodology & per-feature detail vs 9router, OpenRouter, CLIProxyAPI & LiteLLM → [`docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md`](docs/comparison/OMNIROUTE_VS_ALTERNATIVES.md)</sub>
|
||||
|
||||
@@ -517,9 +517,9 @@ Pix copia-e-cola:
|
||||
|
||||
## 📡 OmniRoute Radar
|
||||
|
||||
The main free-tier headline remains **~1.53B tokens/month** from the documented,
|
||||
The main free-tier headline remains **~1.51B tokens/month** from the documented,
|
||||
pool-deduplicated catalog above. Temporary provider signup credits can separately lift the first
|
||||
month to **~2.15B**. Radar is an optional, signed catalog overlay for people who want fresher
|
||||
month to **~2.13B**. Radar is an optional, signed catalog overlay for people who want fresher
|
||||
free-model availability between OmniRoute releases; the community catalog and every existing free
|
||||
feature remain free.
|
||||
|
||||
@@ -548,7 +548,7 @@ the current catalog at **[radar.omniroute.online/planos](https://radar.omniroute
|
||||
- **🗜️ Compression hardening** — default-on inflation guard, Caveman packs for DE / FR / JA + Chinese (wényán), RTK filters for Gradle & .NET. → [Compression](docs/compression/COMPRESSION_ENGINES.md)
|
||||
- **💸 Honest flat-rate cost** — subscription / coding-plan providers read **$0** in cost analytics; budget, quota & routing keep estimating. → [API Reference](docs/reference/API_REFERENCE.md)
|
||||
- **⚖️ Quota-Share routing** — split a shared account's quota fairly across pooled keys, work-conserving so idle slices are lent out. → [Resilience Guide](docs/architecture/RESILIENCE_GUIDE.md)
|
||||
- **🤖 One-command CLI/agent setup** — `setup-*` configures 12+ coding tools; `omniroute run` launches 7 CLIs (Claude Code, Codex, Aider, Goose, OpenCode, Qwen Code, Gemini CLI) with zero config written; `omniroute configure` is an interactive provider+model picker with per-context favorites. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
|
||||
- **🤖 One-command CLI/agent setup** — 12 registered `setup-*` commands; `omniroute run` launches 7 CLIs (Claude Code, Codex, Aider, Goose, OpenCode, Qwen Code, Gemini CLI); `omniroute configure` supports 9 targets with an interactive provider+model picker and per-context favorites. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
|
||||
- **🛰️ Remote mode** — drive a remote OmniRoute with scoped tokens (`connect` / `contexts` / `tokens`) + an `antigravity` OAuth helper for VPS installs. → [Remote Mode](docs/guides/REMOTE-MODE.md)
|
||||
- **🧭 Smarter auto-routing** — `auto/<category>:<tier>` combos, **Fusion** (model panel + judge), task-aware routing, per-request model / mode / USD-budget overrides. → [Auto-Combo](docs/routing/AUTO-COMBO.md)
|
||||
- **🗜️ Pluggable compression** — 12 composable engines + Compression Studios: LLMLingua-2, two-tier Ultra, omniglyph, per-step fidelity gate, GCF v3.2, drag-reorder editor. → [Compression](docs/compression/COMPRESSION_ENGINES.md)
|
||||
@@ -642,11 +642,11 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
|
||||
|
||||
<div align="center">
|
||||
|
||||
## 🌐 349 AI Providers — 90+ Free
|
||||
## 🌐 350 AI Providers — 154 Catalog-Marked Free
|
||||
|
||||
</div>
|
||||
|
||||
> The most complete catalog of any open-source router: **350 providers**, **90+ with a free tier**, **56 free forever**.
|
||||
> **350 registered providers** across the canonical chat, media, search, local, cloud-agent and system collections, including **154 carrying `hasFree: true` discovery metadata**. The chat model registry covers **268 providers / 2,566 distinct provider-model pairs / 1,312 raw model IDs**; the separate free-budget catalog has **455 per-model rows**, **40 recurring pools** and **56 recurring/keyless free-forever providers**. These are different denominators by design; definitions and pool-deduped calculations live in the [Provider Reference](docs/reference/PROVIDER_REFERENCE.md) and [Free Tiers](docs/reference/FREE_TIERS.md).
|
||||
|
||||
<div align="center">
|
||||
|
||||
@@ -679,7 +679,7 @@ of your shell history. → [CLI Integrations](docs/guides/CLI-INTEGRATIONS.md)
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<sub>…and 220+ more — every icon resolves live from the dashboard's provider catalog. 📖 [Provider Reference](docs/reference/PROVIDER_REFERENCE.md)</sub>
|
||||
<sub>…and 330+ more — every icon resolves live from the dashboard's provider catalog. 📖 [Provider Reference](docs/reference/PROVIDER_REFERENCE.md)</sub>
|
||||
|
||||
<br/>
|
||||
|
||||
@@ -769,7 +769,7 @@ From inside the editor: open the **Extensions** view, search **"OmniRoute"**, cl
|
||||
|
||||
</div>
|
||||
|
||||
<img src="./docs/diagrams/privacy-local.svg" width="100%" alt="Private and local-first — your keys, your machine, your data; OmniRoute is a local proxy that never phones home. Eleven guarantees: runs 100% on your hardware (0 cloud hops), zero telemetry by default, credentials encrypted at rest (AES-256-GCM), no account or sign-up, hardened gateway (API-key scoping, IP filtering, rate limits, prompt-injection guard), loopback-only process routes, upstream header scrubbing, strictly opt-in PII redaction, sanitized errors that never leak internals, a local audit trail in your own SQLite, and MIT-licensed fully open-source code."/>
|
||||
<img src="./docs/diagrams/privacy-local.svg" width="100%" alt="Private and local-first — OmniRoute's gateway and control plane run on your machine. Prompts are sent to the upstream provider selected for each request; OmniRoute adds no hosted prompt-processing hop and telemetry is disabled by default. Credentials are encrypted at rest with AES-256-GCM; controls include API-key scoping, IP filtering, rate limits, prompt-injection guards, upstream-header scrubbing, opt-in PII redaction, sanitized errors and a local SQLite audit trail. OmniRoute is MIT-licensed and self-hostable."/>
|
||||
|
||||
<sub>📖 [Authorization](docs/architecture/AUTHZ_GUIDE.md) · [Guardrails](docs/security/GUARDRAILS.md) · [Compliance](docs/security/COMPLIANCE.md)</sub>
|
||||
|
||||
@@ -810,7 +810,7 @@ Tokens are scoped `read` / `write` / `admin`; process-spawning routes stay loopb
|
||||
|
||||
<div align="left">
|
||||
|
||||
<img src="./docs/diagrams/cli-terminal.svg" width="50%" alt="Animated terminal demoing the OmniRoute CLI — omniroute providers list, omniroute combo list, omniroute health — cycling over the 80+ command surface: providers · oauth · keys · combo · nodes · models · cache · compression · cost · usage · quota · health · resilience · telemetry · logs · audit · mcp · a2a · cloud · memory · skills · eval · tunnel · backup · sync · webhooks · policy · pricing · translator · simulate …"/>
|
||||
<img src="./docs/diagrams/cli-terminal.svg" width="50%" alt="Animated terminal demoing the OmniRoute CLI — omniroute providers list, omniroute combo list and omniroute health — cycling over the 85-command top-level surface: providers · oauth · keys · combo · nodes · models · cache · compression · cost · usage · quota · health · resilience · telemetry · logs · audit · mcp · a2a · cloud · memory · skills · eval · tunnel · backup · sync · webhooks · policy · pricing · translator · simulate …"/>
|
||||
|
||||
</div>
|
||||
|
||||
@@ -846,7 +846,7 @@ claude mcp add-server omniroute --type http --url http://localhost:20128/api/mcp
|
||||
|
||||
### 📖 How it works — pipeline, architecture & savings math
|
||||
|
||||
<img src="./docs/diagrams/compression-pipeline.svg" width="100%" alt="OmniRoute compression pipeline: a client request of 10,000 tokens passes through 12 stacked engines — Session-Dedup, CCR, Lite, RTK, Responses Tool Output, Headroom, Relevance, Caveman, Aggressive, LLMLingua-2, Ultra, OmniGlyph — and reaches the provider at about 1,080 tokens, up to 95% saved. Code, URLs and JSON are always preserved byte-perfect."/>
|
||||
<img src="./docs/diagrams/compression-pipeline.svg" width="100%" alt="OmniRoute compression pipeline: an illustrative 10,000-token client request passes through 12 composable engines — Session-Dedup, CCR, Lite, RTK, Responses Tool Output, Headroom, Relevance, Caveman, Aggressive, LLMLingua-2, Ultra and OmniGlyph — and can reach the provider at about 1,080 tokens in the documented stacked example. Structured content is protected by preservation guards and per-step fidelity gates; explicit lossy or experimental modes may transform eligible content."/>
|
||||
|
||||
Default stacked combo runs `RTK → Caveman`. When both act on the same tool/context payload, savings compound:
|
||||
|
||||
@@ -1013,6 +1013,7 @@ Full table: [Docker Guide — runtime RAM](docs/guides/DOCKER_GUIDE.md#runtime-r
|
||||
**🥟 Bun**
|
||||
|
||||
Standard `bun install` and global installation (`bun install -g omniroute`) are supported via Bun runtime detection:
|
||||
|
||||
- **Built-in `bun:sqlite`**: OmniRoute uses Bun's built-in `bun:sqlite` driver when running under Bun, falling back to `better-sqlite3` on Node.js or `sql.js`.
|
||||
- **Automatic Webpack bundler selection**: Development (`bun run dev`) and production builds (`bun run build`) automatically detect Bun and disable Turbopack in favor of Webpack to prevent native V8 binding incompatibilities.
|
||||
- **Dedicated Bun Dockerfile**: Multi-stage `Dockerfile.bun` for native Bun production deployments (`docker build -f Dockerfile.bun -t omniroute:bun .`).
|
||||
@@ -1105,7 +1106,7 @@ same process on one port, so there is no separate CLI-only package today.
|
||||
|
||||
<div align="center">
|
||||
|
||||
<sub>Dados de cobertura social em 2026-08-17 · YT: 741 | TT: 137 | IG: 124 · Frescor (dias): YT 0 · TT 14 · IG 15</sub>
|
||||
<sub>Snapshot do painel em 2026-08-24 · Catálogo bruto: YT 809 | TT 137 | IG 124 · Frescor (dias): YT 1 | TT 21 | IG 22</sub>
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
@@ -1114,52 +1115,52 @@ same process on one port, so there is no separate CLI-only package today.
|
||||
<img src="https://placehold.co/320x180/111827/FFFFFF?text=Instagram+Reel+%7C+nick_saraev&font=montserrat&bold=true" alt="Instagram Reel" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #1 — Instagram</b><br/>
|
||||
<sub>nick_saraev — 1,628,910 views</sub>
|
||||
<sub>nick_saraev — 3,042,474 views</sub>
|
||||
</td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.instagram.com/reel/DaSs65mMrHk/">
|
||||
<img src="https://placehold.co/320x180/111827/FFFFFF?text=Instagram+Reel+%7C+theopenstack&font=montserrat&bold=true" alt="Instagram Reel — theopenstack" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #2 — Instagram</b><br/>
|
||||
<sub>theopenstack — 692,419 views</sub>
|
||||
</td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.tiktok.com/@milesreevesai/video/7667980059189366019">
|
||||
<img src="https://placehold.co/320x180/111827/FFFFFF?text=TikTok+%7C+milesreevesai&font=montserrat&bold=true" alt="TikTok — milesreevesai" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #3 — TikTok</b><br/>
|
||||
<sub>milesreevesai — 620,400 views</sub>
|
||||
</td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.youtube.com/watch?v=QucgvbO5gsM">
|
||||
<img src="https://img.youtube.com/vi/QucgvbO5gsM/maxresdefault.jpg" alt="YouTube — Vaibhav Sisinty" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #2 — YouTube</b><br/>
|
||||
<sub>Vaibhav Sisinty — 373,084 views</sub>
|
||||
<b>🎬 #4 — YouTube</b><br/>
|
||||
<sub>Vaibhav Sisinty — 391,109 views</sub>
|
||||
</td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.youtube.com/shorts/fZIBK_4fKq8">
|
||||
<img src="https://img.youtube.com/vi/fZIBK_4fKq8/maxresdefault.jpg" alt="YouTube Shorts" width="300"/>
|
||||
<a href="https://www.instagram.com/reel/DbIt9AjK7-U/">
|
||||
<img src="https://placehold.co/320x180/111827/FFFFFF?text=Instagram+Reel+%7C+buildwithai.club&font=montserrat&bold=true" alt="Instagram Reel — buildwithai.club" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #3 — YouTube Shorts</b><br/>
|
||||
<sub>Nick Automates — 207,714 views</sub>
|
||||
</td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.tiktok.com/@milesreevesai/video/7667980059189366019">
|
||||
<img src="https://placehold.co/320x180/111827/FFFFFF?text=TikTok+Top+1&font=montserrat&bold=true" alt="TikTok Thumbnail" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #4 — TikTok</b><br/>
|
||||
<sub>milesreevesai — 620,400 views</sub>
|
||||
</td>
|
||||
<td align="center" width="320">
|
||||
<a href="https://www.youtube.com/watch?v=LkP6ocAoQkk">
|
||||
<img src="https://img.youtube.com/vi/LkP6ocAoQkk/maxresdefault.jpg" alt="Valency Labs" width="300"/>
|
||||
</a><br/>
|
||||
<b>🎬 #5 — YouTube</b><br/>
|
||||
<sub>Valency Labs — 135,974 views</sub>
|
||||
<b>🎬 #5 — Instagram</b><br/>
|
||||
<sub>buildwithai.club — 347,652 views</sub>
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
</div>
|
||||
|
||||
**Ranking completo (`v > 0`, maior alcance):**
|
||||
**Ranking completo (URLs canônicas deduplicadas, `v > 0`, maior alcance):**
|
||||
|
||||
| #1 | #2 | #3 | #4 | #5 |
|
||||
| -------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| [nick_saraev — Instagram](https://www.instagram.com/reel/Da8ZthUPK98/) — **1,628,910** | [milesreevesai — TikTok](https://www.tiktok.com/@milesreevesai/video/7667980059189366019) — **620,400** | [Vaibhav Sisinty — YouTube](https://www.youtube.com/watch?v=QucgvbO5gsM) — **373,084** | [Nick Automates — YouTube Shorts](https://www.youtube.com/shorts/fZIBK_4fKq8) — **207,714** | [midudev — TikTok](https://www.tiktok.com/@midudev/video/7664636453544152342) — **177,800** |
|
||||
| #1 | #2 | #3 | #4 | #5 |
|
||||
| -------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------- |
|
||||
| [nick_saraev — Instagram](https://www.instagram.com/reel/Da8ZthUPK98/) — **3,042,474** | [theopenstack — Instagram](https://www.instagram.com/reel/DaSs65mMrHk/) — **692,419** | [milesreevesai — TikTok](https://www.tiktok.com/@milesreevesai/video/7667980059189366019) — **620,400** | [Vaibhav Sisinty — YouTube](https://www.youtube.com/watch?v=QucgvbO5gsM) — **391,109** | [buildwithai.club — Instagram](https://www.instagram.com/reel/DbIt9AjK7-U/) — **347,652** |
|
||||
|
||||
| #6 | #7 | #8 | #9 | #10 |
|
||||
| ------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------- | ---------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- |
|
||||
| [theopenstack — Instagram](https://www.instagram.com/reel/DaSs65mMrHk/) — **155,453** | [t.ghoush.ai — TikTok](https://www.tiktok.com/@t.ghoush.ai/video/7669497680527248656) — **152,800** | [Valency Labs — YouTube](https://www.youtube.com/watch?v=LkP6ocAoQkk) — **135,974** | [Asati — YouTube](https://www.youtube.com/watch?v=JjPtJcqwhqg) — **126,130** | [Vaibhav Sisinty — YouTube](https://www.youtube.com/watch?v=NuNDpeZYQ28) — **122,672** |
|
||||
| #6 | #7 | #8 | #9 | #10 |
|
||||
| ----------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| [nivedan.ai — Instagram](https://www.instagram.com/reel/DbIrCksJiqq/) — **331,973** | [vaibhavsisinty — Instagram](https://www.instagram.com/reel/Dae05TSAK1l/) — **263,744** | [Nick Automates — YouTube Shorts](https://www.youtube.com/shorts/fZIBK_4fKq8) — **218,174** | [theroshankrishna — Instagram](https://www.instagram.com/reel/Dapjs58z0P0/) — **186,786** | [midudev — TikTok](https://www.tiktok.com/@midudev/video/7664636453544152342) — **177,800** |
|
||||
|
||||
Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações conhecidas · 595 perfis/canais · 13+ idiomas · 13+ criadores.
|
||||
Métricas canônicas em 2026-08-24: **1.029 vídeos únicos** · **11.132.922 visualizações conhecidas** (`v > 0`) · **639 canais/perfis por rede**. O painel bruto contém 1.070 linhas; 41 duplicatas do Instagram foram normalizadas pela URL canônica, mantendo a maior contagem por vídeo.
|
||||
|
||||
> 🎬 **Made a video about OmniRoute?** Open an [issue](https://github.com/diegosouzapw/OmniRoute/issues/new) or [discussion](https://github.com/diegosouzapw/OmniRoute/discussions) with the link — we'll feature it here.
|
||||
|
||||
@@ -1211,7 +1212,7 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c
|
||||
<tr><td nowrap><b>Stealth</b></td><td>wreq-js — JA3 / JA4 TLS fingerprint impersonation, 3-level proxy</td></tr>
|
||||
<tr><td nowrap><b>Resilience</b></td><td>Circuit breaker, exponential backoff, anti-thundering-herd, auto-combo self-healing</td></tr>
|
||||
<tr><td nowrap><b>Logging</b></td><td>pino — structured JSON logs with request context</td></tr>
|
||||
<tr><td nowrap><b>Testing</b></td><td>Node.js test runner + Vitest — <b>25,000+ test cases</b> across 3,300+ files (unit, integration, E2E, security, ecosystem)</td></tr>
|
||||
<tr><td nowrap><b>Testing</b></td><td>Node.js test runner + Vitest — <b>39,000+ static test declarations</b> across 5,100+ tracked test files (unit, integration, E2E, security, ecosystem)</td></tr>
|
||||
<tr><td nowrap><b>Platforms</b></td><td>Desktop (Electron) · Android (Termux) · PWA (any browser)</td></tr>
|
||||
<tr><td nowrap><b>CI/CD</b></td><td>GitHub Actions — auto npm publish + Docker Hub on release</td></tr>
|
||||
<tr><td nowrap><b>Links</b></td><td><a href="https://omniroute.online">Website</a> · <a href="https://www.npmjs.com/package/omniroute">npm</a> · <a href="https://hub.docker.com/r/diegosouzapw/omniroute">Docker Hub</a></td></tr>
|
||||
@@ -1262,9 +1263,9 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c
|
||||
<tr><td nowrap><b><a href="docs/compression/COMPRESSION_RULES_FORMAT.md">Compression Rules Format</a></b></td><td>JSON rule-pack schemas for Caveman and RTK filters</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/compression/COMPRESSION_LANGUAGE_PACKS.md">Compression Language Packs</a></b></td><td>Language detection and Caveman rule-pack authoring</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/architecture/RESILIENCE_GUIDE.md">Resilience Guide</a></b></td><td>Circuit breakers, cooldowns, queue, anti-thundering herd, TLS spoofing</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/routing/AUTO-COMBO.md">Auto-Combo Engine</a></b></td><td>14-factor scoring, mode packs, self-healing</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/routing/AUTO-COMBO.md">Auto-Combo Engine</a></b></td><td>15-factor scoring, mode packs, self-healing</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/ops/PROXY_GUIDE.md">Proxy Guide</a></b></td><td>3-level proxy system, 1proxy marketplace, registry CRUD</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/reference/FREE_TIERS.md">Free Tiers</a></b></td><td>90+ free providers consolidated directory (42 documented token pools / 495 models)</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/reference/FREE_TIERS.md">Free Tiers</a></b></td><td>Consolidated directory: 40 documented recurring pools / 455 cataloged free-tier entries</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/guides/FEATURES.md">Features Gallery</a></b></td><td>Visual dashboard tour with screenshots</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/architecture/CODEBASE_DOCUMENTATION.md">Codebase Documentation</a></b></td><td>Beginner-friendly codebase walkthrough</td></tr>
|
||||
</table>
|
||||
@@ -1275,7 +1276,7 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c
|
||||
<tr><th align="left">Document</th><th align="left">Description</th></tr>
|
||||
<tr><td nowrap><b><a href="docs/reference/API_REFERENCE.md">API Reference</a></b></td><td>All endpoints with examples</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/openapi.yaml">OpenAPI Spec</a></b></td><td>OpenAPI 3.0 specification</td></tr>
|
||||
<tr><td nowrap><b><a href="open-sse/mcp-server/README.md">MCP Server</a></b></td><td>109 MCP tools, IDE configs, Python/TS/Go clients</td></tr>
|
||||
<tr><td nowrap><b><a href="open-sse/mcp-server/README.md">MCP Server</a></b></td><td>110 MCP tools, IDE configs, Python/TS/Go clients</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/frameworks/MCP-SERVER.md">MCP Server Guide</a></b></td><td>MCP installation, transports, and tool reference</td></tr>
|
||||
<tr><td nowrap><b><a href="src/lib/a2a/README.md">A2A Server</a></b></td><td>JSON-RPC 2.0 protocol, skills, streaming, task mgmt</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/frameworks/A2A-SERVER.md">A2A Server Guide</a></b></td><td>A2A agent card, tasks, skills, and streaming</td></tr>
|
||||
@@ -1291,7 +1292,7 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c
|
||||
<tr><td nowrap><b><a href="SECURITY.md">Security Policy</a></b></td><td>Vulnerability reporting and security practices</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/guides/I18N.md">i18n Guide</a></b></td><td>43-language support, translation workflow, RTL</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/ops/RELEASE_CHECKLIST.md">Release Checklist</a></b></td><td>Pre-release validation steps</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/ops/COVERAGE_PLAN.md">Coverage Plan</a></b></td><td>Test coverage strategy and 25,000+ test suite</td></tr>
|
||||
<tr><td nowrap><b><a href="docs/ops/COVERAGE_PLAN.md">Coverage Plan</a></b></td><td>Test coverage strategy for 39,000+ static test declarations across 5,100+ tracked test files</td></tr>
|
||||
</table>
|
||||
|
||||
<br/>
|
||||
@@ -1302,93 +1303,123 @@ Métricas de validação: 1002 vídeos rastreados · 7,069,190 visualizações c
|
||||
|
||||
> OmniRoute is shaped by a passionate open-source community. These individuals have made exceptional contributions that directly impact the quality, stability, and reach of the project. **Thank you.**
|
||||
|
||||
### External contributors by merged pull requests
|
||||
|
||||
<table>
|
||||
<tr><th align="center">Rank</th><th align="left">Contributor</th><th align="center">Merged PRs</th><th align="right">~Changed lines</th></tr>
|
||||
<tr><td align="center">1</td><td align="left"><a href="https://github.com/backryun"><b>backryun</b></a></td><td align="center">190</td><td align="right">227,977</td></tr>
|
||||
<tr><td align="center">2</td><td align="left"><a href="https://github.com/oyi77"><b>oyi77</b></a></td><td align="center">180</td><td align="right">407,678</td></tr>
|
||||
<tr><td align="center">3</td><td align="left"><a href="https://github.com/rdself"><b>rdself</b></a></td><td align="center">145</td><td align="right">80,663</td></tr>
|
||||
<tr><td align="center">4</td><td align="left"><a href="https://github.com/JxnLexn"><b>JxnLexn</b></a></td><td align="center">128</td><td align="right">387,049</td></tr>
|
||||
<tr><td align="center">5</td><td align="left"><a href="https://github.com/KooshaPari"><b>KooshaPari</b></a></td><td align="center">101</td><td align="right">125,747</td></tr>
|
||||
<tr><td align="center">6</td><td align="left"><a href="https://github.com/herjarsa"><b>herjarsa</b></a></td><td align="center">88</td><td align="right">230,872</td></tr>
|
||||
<tr><td align="center">7</td><td align="left"><a href="https://github.com/RaviTharuma"><b>RaviTharuma</b></a></td><td align="center">79</td><td align="right">55,106</td></tr>
|
||||
<tr><td align="center">8</td><td align="left"><a href="https://github.com/maxmad64bis"><b>maxmad64bis</b></a></td><td align="center">69</td><td align="right">394,715</td></tr>
|
||||
<tr><td align="center">9</td><td align="left"><a href="https://github.com/artickc"><b>artickc</b></a></td><td align="center">59</td><td align="right">33,260</td></tr>
|
||||
<tr><td align="center">10</td><td align="left"><a href="https://github.com/HouMinXi"><b>HouMinXi</b></a></td><td align="center">51</td><td align="right">47,334</td></tr>
|
||||
<tr><td align="center">10</td><td align="left"><a href="https://github.com/chirag127"><b>chirag127</b></a></td><td align="center">51</td><td align="right">5,153</td></tr>
|
||||
<tr><td align="center">12</td><td align="left"><a href="https://github.com/xz-dev"><b>xz-dev</b></a></td><td align="center">50</td><td align="right">245,976</td></tr>
|
||||
<tr><td align="center">13</td><td align="left"><a href="https://github.com/hartmark"><b>hartmark</b></a></td><td align="center">47</td><td align="right">52,185</td></tr>
|
||||
<tr><td align="center">14</td><td align="left"><a href="https://github.com/rqzbeh"><b>rqzbeh</b></a></td><td align="center">39</td><td align="right">143,181</td></tr>
|
||||
<tr><td align="center">15</td><td align="left"><a href="https://github.com/dhaern"><b>dhaern</b></a></td><td align="center">34</td><td align="right">19,559</td></tr>
|
||||
<tr><td align="center">16</td><td align="left"><a href="https://github.com/Dingding-leo"><b>Dingding-leo</b></a></td><td align="center">33</td><td align="right">1,986</td></tr>
|
||||
<tr><td align="center">17</td><td align="left"><a href="https://github.com/NomenAK"><b>NomenAK</b></a></td><td align="center">32</td><td align="right">13,854</td></tr>
|
||||
<tr><td align="center">18</td><td align="left"><a href="https://github.com/MumuTW"><b>MumuTW</b></a></td><td align="center">30</td><td align="right">16,953</td></tr>
|
||||
<tr><td align="center">19</td><td align="left"><a href="https://github.com/benzntech"><b>benzntech</b></a></td><td align="center">29</td><td align="right">11,641</td></tr>
|
||||
<tr><td align="center">20</td><td align="left"><a href="https://github.com/pacocartones"><b>pacocartones</b></a></td><td align="center">24</td><td align="right">9,331</td></tr>
|
||||
<tr><td align="center">20</td><td align="left"><a href="https://github.com/Prudhvivuda"><b>Prudhvivuda</b></a></td><td align="center">24</td><td align="right">6,312</td></tr>
|
||||
</table>
|
||||
|
||||
<sub>Frozen at live <code>release/v3.8.50</code> tip <code>dafb4ae808</code>, with merges through 2026-08-24 05:26:03 UTC. The paginated GitHub GraphQL census contains 5,911 merged PRs: 2,707 by the repository owner, 179 by Dependabot, and <b>3,025 external PRs from 535 distinct contributors</b>. “Changed lines” is GitHub additions + deletions and includes generated files, lockfiles, catalogs, translations and documentation; it is churn, not authored LOC. Ties at the cutoff are retained.</sub>
|
||||
|
||||
### GitHub-attributed commits
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/oyi77">
|
||||
<img src="https://github.com/oyi77.png" width="40" style="border-radius:50%" alt="oyi77"/><br/>
|
||||
<b>oyi77</b>
|
||||
</a><br/>
|
||||
<sub>🥇 213 commits • +114K lines</sub><br/>
|
||||
<sub>Analytics engine, SQL aggregations,<br/>proxy marketplace, test coverage</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/rdself">
|
||||
<img src="https://github.com/rdself.png" width="40" style="border-radius:50%" alt="R.D. & Randi"/><br/>
|
||||
<b>R.D. & Randi</b>
|
||||
</a><br/>
|
||||
<sub>🥈 108 commits • +38K lines</sub><br/>
|
||||
<sub>Endpoints page, tunnel integrations,<br/>Docker workflows, A2A status, compression UI</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/christopher-s">
|
||||
<img src="https://github.com/christopher-s.png" width="40" style="border-radius:50%" alt="Chris Staley"/><br/>
|
||||
<b>Chris Staley</b>
|
||||
</a><br/>
|
||||
<sub>🥉 70 commits • +1.8K lines</sub><br/>
|
||||
<sub>SSE stream hardening, Responses API,<br/>Gemini pagination, test regression fixes</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/zen0bit">
|
||||
<img src="https://github.com/zen0bit.png" width="40" style="border-radius:50%" alt="zenobit"/><br/>
|
||||
<b>zenobit</b>
|
||||
</a><br/>
|
||||
<sub>🏅 62 commits • +22K lines</sub><br/>
|
||||
<sub>CI/CD pipeline, i18n for 33 languages,<br/>Void Linux package, platform fixes</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/JxnLexn">
|
||||
<img src="https://github.com/JxnLexn.png" width="40" style="border-radius:50%" alt="Jan Leon"/><br/>
|
||||
<b>Jan Leon</b>
|
||||
</a><br/>
|
||||
<sub>🏅 58 commits • +22K lines</sub><br/>
|
||||
<sub>Reasoning-effort routing, proxy controls,<br/>quota visibility, Live Zone compression</sub>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/backryun">
|
||||
<img src="https://github.com/backryun.png" width="40" style="border-radius:50%" alt="backryun"/><br/>
|
||||
<b>backryun</b>
|
||||
</a><br/>
|
||||
<sub>🏅 53 commits • +70K lines</sub><br/>
|
||||
<sub>Provider catalog curation — Perplexity, Kimi,<br/>Cerebras, Copilot, LMArena refreshes</sub>
|
||||
<sub>🥇 220 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/chirag127">
|
||||
<img src="https://github.com/chirag127.png" width="40" style="border-radius:50%" alt="Chirag Singhal"/><br/>
|
||||
<b>Chirag Singhal</b>
|
||||
<a href="https://github.com/oyi77">
|
||||
<img src="https://github.com/oyi77.png" width="40" style="border-radius:50%" alt="Paijo"/><br/>
|
||||
<b>Paijo</b>
|
||||
</a><br/>
|
||||
<sub>🏅 46 commits • +4.8K lines</sub><br/>
|
||||
<sub>Error sanitization, MITM prefill fix,<br/>fusion judge, breaker/429 correctness</sub>
|
||||
<sub>🥈 219 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/kfiramar">
|
||||
<img src="https://github.com/kfiramar.png" width="40" style="border-radius:50%" alt="kfiramar"/><br/>
|
||||
<b>kfiramar</b>
|
||||
<a href="https://github.com/rdself">
|
||||
<img src="https://github.com/rdself.png" width="40" style="border-radius:50%" alt="Randi"/><br/>
|
||||
<b>Randi</b>
|
||||
</a><br/>
|
||||
<sub>🏅 38 commits • +1.7K lines</sub><br/>
|
||||
<sub>Codex websocket + passthrough, auth/onboarding,<br/>Electron hardening, DB migrations</sub>
|
||||
<sub>🥉 108 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/benzntech">
|
||||
<img src="https://github.com/benzntech.png" width="40" style="border-radius:50%" alt="Benson K B"/><br/>
|
||||
<b>Benson K B</b>
|
||||
<a href="https://github.com/RaviTharuma">
|
||||
<img src="https://github.com/RaviTharuma.png" width="40" style="border-radius:50%" alt="Ravi Tharuma"/><br/>
|
||||
<b>Ravi Tharuma</b>
|
||||
</a><br/>
|
||||
<sub>🏅 28 commits • +9.2K lines</sub><br/>
|
||||
<sub>Electron desktop app, auto-updater,<br/>release build workflows, cross-platform CI</sub>
|
||||
<sub>🏅 81 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/herjarsa">
|
||||
<img src="https://github.com/herjarsa.png" width="40" style="border-radius:50%" alt="Hernan J. Ardila"/><br/>
|
||||
<b>Hernan J. Ardila</b>
|
||||
<a href="https://github.com/christopher-s">
|
||||
<img src="https://github.com/christopher-s.png" width="40" style="border-radius:50%" alt="Chris"/><br/>
|
||||
<b>Chris</b>
|
||||
</a><br/>
|
||||
<sub>🏅 25 commits • +174K lines</sub><br/>
|
||||
<sub>Zero-latency combos, vision-bridge auto-routing,<br/>catalog context-length, resilience 429 hints</sub>
|
||||
<sub>🏅 70 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/hartmark">
|
||||
<img src="https://github.com/hartmark.png" width="40" style="border-radius:50%" alt="Markus Hartung"/><br/>
|
||||
<b>Markus Hartung</b>
|
||||
</a><br/>
|
||||
<sub>🏅 69 GitHub-attributed commits · tied #6</sub>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/maxmad64bis">
|
||||
<img src="https://github.com/maxmad64bis.png" width="40" style="border-radius:50%" alt="Dizzle"/><br/>
|
||||
<b>Dizzle</b>
|
||||
</a><br/>
|
||||
<sub>🏅 69 GitHub-attributed commits · tied #6</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/JxnLexn">
|
||||
<img src="https://github.com/JxnLexn.png" width="40" style="border-radius:50%" alt="Jan Leon"/><br/>
|
||||
<b>Jan Leon</b>
|
||||
</a><br/>
|
||||
<sub>🏅 64 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/zen0bit">
|
||||
<img src="https://github.com/zen0bit.png" width="40" style="border-radius:50%" alt="zenobit"/><br/>
|
||||
<b>zenobit</b>
|
||||
</a><br/>
|
||||
<sub>🏅 62 GitHub-attributed commits</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/HouMinXi">
|
||||
<img src="https://github.com/HouMinXi.png" width="40" style="border-radius:50%" alt="Bob.Hou"/><br/>
|
||||
<b>Bob.Hou</b>
|
||||
</a><br/>
|
||||
<sub>🏅 51 GitHub-attributed commits · tied #10</sub>
|
||||
</td>
|
||||
<td align="center" width="160">
|
||||
<a href="https://github.com/xz-dev">
|
||||
<img src="https://github.com/xz-dev.png" width="40" style="border-radius:50%" alt="Xiangzhe"/><br/>
|
||||
<b>Xiangzhe</b>
|
||||
</a><br/>
|
||||
<sub>🏅 51 GitHub-attributed commits · tied #10</sub>
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<sub>Rechecked at 2026-08-24 06:14:31 UTC: GitHub-attributed commits reported by the repository Contributors API for the <code>release/v3.8.50</code> default branch. The API returned 525 identities (415 users, 2 bots, 108 anonymous); this table excludes the maintainer, bots and anonymous identities and retains competition ties. It is distinct from both the merged-PR ranking above and the 639-person Git-metadata census below.</sub>
|
||||
|
||||
> 🙏 These contributors' features, bug fixes, and infrastructure improvements are a **core part** of what makes OmniRoute reliable and feature-rich. Every pull request, every test case, and every i18n translation file matters. Open source is built by people like them.
|
||||
|
||||
</div>
|
||||
@@ -1405,25 +1436,48 @@ A heartfelt thank-you to the people who fund OmniRoute out of their own pocket
|
||||
|
||||
<table>
|
||||
<tr>
|
||||
<td align="center" width="180">
|
||||
<a href="https://github.com/drewbitt">
|
||||
<img src="https://github.com/drewbitt.png?size=140" width="72" style="border-radius:50%" alt="Andrew"/><br/>
|
||||
<b>Andrew</b>
|
||||
</a><br/>
|
||||
<sub>💛 Active monthly sponsor</sub>
|
||||
</td>
|
||||
<td align="center" width="180">
|
||||
<a href="https://github.com/psylligent">
|
||||
<img src="https://github.com/psylligent.png?size=140" width="72" style="border-radius:50%" alt="Vlad I"/><br/>
|
||||
<b>Vlad I</b>
|
||||
</a><br/>
|
||||
<sub>💛 Active monthly sponsor</sub>
|
||||
</td>
|
||||
<td align="center" width="180">
|
||||
<a href="https://github.com/pacocartones">
|
||||
<img src="https://github.com/pacocartones.png?size=140" width="72" style="border-radius:50%" alt="Paco Cartones"/><br/>
|
||||
<b>Paco Cartones</b>
|
||||
</a><br/>
|
||||
<sub>💛 Active one-time sponsor</sub>
|
||||
</td>
|
||||
<td align="center" width="180">
|
||||
<a href="https://github.com/igormorais123">
|
||||
<img src="https://github.com/igormorais123.png?size=140" width="72" style="border-radius:50%" alt="Professor Igor Morais Vasconcelos"/><br/>
|
||||
<b>Prof. Igor Morais</b>
|
||||
</a><br/>
|
||||
<sub>💛 Sponsor</sub>
|
||||
<sub>💛 Past one-time supporter</sub>
|
||||
</td>
|
||||
<td align="center" width="180">
|
||||
<a href="https://github.com/longtao77">
|
||||
<img src="https://github.com/longtao77.png?size=140" width="72" style="border-radius:50%" alt="longtao"/><br/>
|
||||
<b>longtao</b>
|
||||
</a><br/>
|
||||
<sub>💛 Sponsor</sub>
|
||||
<sub>💛 Past one-time supporter</sub>
|
||||
</td>
|
||||
</tr>
|
||||
</table>
|
||||
|
||||
<sub>… and others who prefer to stay private 💛</sub>
|
||||
|
||||
<sub>Public GitHub Sponsors revalidated on 2026-08-24. GitHub's <code>activeOnly</code> status determines the active labels above; previously disclosed public one-time supporters remain thanked, and private sponsors remain anonymous.</sub>
|
||||
|
||||
<b><a href="https://github.com/sponsors/diegosouzapw">💖 Become a sponsor →</a></b> — every dollar keeps OmniRoute free and independent.
|
||||
|
||||
</div>
|
||||
@@ -1432,11 +1486,13 @@ A heartfelt thank-you to the people who fund OmniRoute out of their own pocket
|
||||
|
||||
<div align="center">
|
||||
|
||||
## 👥 320+ Contributors
|
||||
## 👥 600+ Contributors
|
||||
|
||||
</div>
|
||||
|
||||
[](https://github.com/diegosouzapw/OmniRoute/graphs/contributors)
|
||||
[](https://github.com/diegosouzapw/OmniRoute/graphs/contributors)
|
||||
|
||||
<sub>Audited on 2026-08-24 at frozen base <code>ac02c5b42f</code> and rechecked at live <code>release/v3.8.50</code> tip <code>dafb4ae808</code>: <b>639 normalized human Git identities</b> — 407 appear as commit authors (including the maintainer) and 232 only in explicit <code>Co-authored-by</code> trailers. The census normalizes GitHub noreply handles, excludes 26 bot/agent/service/placeholder identities, and does not merge ordinary email addresses merely because their display names match.</sub>
|
||||
|
||||
### How to Contribute
|
||||
|
||||
@@ -1453,7 +1509,8 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) for detailed guidelines.
|
||||
|
||||
```bash
|
||||
# Create a release — npm publish happens automatically
|
||||
gh release create v3.8.2 --title "v3.8.2" --generate-notes
|
||||
VERSION=x.y.z
|
||||
gh release create "v${VERSION}" --title "v${VERSION}" --generate-notes
|
||||
```
|
||||
|
||||
<br/>
|
||||
@@ -1495,88 +1552,108 @@ gh release create v3.8.2 --title "v3.8.2" --generate-notes
|
||||
|
||||
OmniRoute stands on the shoulders of giants. It started as a fork of **[9router](https://github.com/decolua/9router)** and a TypeScript port of the Go project **[CLIProxyAPI](https://github.com/router-for-me/CLIProxyAPI)** — and from there, every subsystem below was inspired by an open-source project that got there first. Each one shaped a concrete piece of OmniRoute. This is our thank-you to all of them. 🙏
|
||||
|
||||
> ⭐ star counts as of July 2026 — go give these projects a star.
|
||||
> ⭐ star counts verified from GitHub's REST API on August 24, 2026 — go give these projects a star. Counts are an exact dated snapshot and will naturally change.
|
||||
|
||||
### 🧬 Lineage & gateway
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/decolua/9router">9router</a></b></td><td align="center">22.7k</td><td>The original project this fork is built on — extended here with multi-modal APIs and a full TypeScript rewrite.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/router-for-me/CLIProxyAPI">CLIProxyAPI</a></b></td><td align="center">43.6k</td><td>The Go implementation that inspired this JavaScript / TypeScript port.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/BerriAI/litellm">LiteLLM</a></b></td><td align="center">54.0k</td><td>The AI gateway whose public pricing dataset feeds our cost-tracking sync and whose provider-normalization model informed our routing.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/decolua/9router">9router</a></b></td><td align="center">26,161</td><td>The original project this fork is built on — extended here with multi-modal APIs and a full TypeScript rewrite.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/router-for-me/CLIProxyAPI">CLIProxyAPI</a></b></td><td align="center">48,497</td><td>The Go implementation that inspired this JavaScript / TypeScript port.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/BerriAI/litellm">LiteLLM</a></b></td><td align="center">57,100</td><td>The AI gateway whose public pricing dataset feeds our cost-tracking sync and whose provider-normalization model informed our routing.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/miuuyy/codex-chatgpt-web">codex-chatgpt-web</a></b></td><td align="center">1,410</td><td>MIT source adapted into the vendored ChatGPT Web → Codex Responses bridge, including browser-session, response-framing, usage and web-search adapters.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/Alishahryar1/free-claude-code">free-claude-code</a></b></td><td align="center">48,112</td><td>Patterns ported into stream recovery, no-thinking aliases, fallback web search, sliding-window limits, log redaction and hardened launcher flows.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/standardagents/composer-api">composer-api</a></b></td><td align="center">322</td><td>Cursor Composer tool-choice, output-constraint and tool-commit patterns adapted into the native Cursor executor.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ndycode/codex-multi-auth">codex-multi-auth</a></b></td><td align="center">457</td><td>Fresh-login and refresh-token rotation patterns ported into Codex OAuth reauthentication.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ex-machina-co/opencode-anthropic-auth">opencode-anthropic-auth</a></b></td><td align="center">510</td><td>Claude Code-compatible transform defaults and billing-header behavior generalized into OmniRoute's config-driven bridge.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/520mmxx/grok2api-merged">grok2api-merged</a></b></td><td align="center">2</td><td>Its Grok model mappings, fake-TypeError Statsig generator, request and device defaults, and NDJSON response processor were materially adapted into OmniRoute's Grok Web executor.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/TQZHR/grok2api">TQZHR/grok2api</a></b></td><td align="center">705</td><td>The principal transitive code source behind grok2api-merged; its model, header, payload, Statsig and processor implementations are preserved in the Grok Web lineage.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/chenyme/grok2api">chenyme/grok2api</a></b></td><td align="center">7,520</td><td>The underlying MIT source for Grok payload and device defaults, the Statsig generator, and the <code>result.response</code> processor carried through TQZHR and grok2api-merged.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/miuzhaii/grok2api-pro">grok2api-pro</a></b></td><td align="center">27</td><td>A transitive source credited by grok2api-merged for its proxy-pool layer; OmniRoute preserves that lineage notice but does not claim a proxy-pool port in its bounded Grok Web executor.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/CNFlyCat/GrokProxy">GrokProxy</a></b></td><td align="center">50</td><td>Its cookie-authenticated Grok proxy and <code>result.response.token</code> streaming pattern informed OmniRoute's Grok Web transport.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/lianying1716/GrokBridge">GrokBridge</a></b></td><td align="center">5</td><td>The original Grok Web implementation consulted its HTTP/browser upstream design; its direct HTTP path derives from GrokProxy, so no independent code port is claimed.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/imjustprism/grok-web-api">grok-web-api</a></b></td><td align="center">14</td><td>Its Rust <code>ChatOptions</code> and response-envelope schemas informed OmniRoute's TypeScript Grok request and streaming-response types.</td></tr>
|
||||
</table>
|
||||
|
||||
### 🗜️ Context & token compression — engines
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/JuliusBrussee/caveman">Caveman</a></b></td><td align="center">90.8k</td><td>The viral "why use many token when few token do trick" project — its caveman-speak philosophy powers our standard compression mode and 30+ filler/condensation rules.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/rtk-ai/rtk">RTK – Rust Token Killer</a></b></td><td align="center">71.8k</td><td>High-performance command-output compression — inspired our RTK engine, JSON filter DSL, raw-output recovery and the stacked RTK → Caveman pipeline.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/headroomlabs-ai/headroom">headroom</a></b></td><td align="center">60.1k</td><td>Reversible context-compression (SmartCrusher) — inspired our <code>headroom</code> engine and the <code>ccr</code> retrieve-marker pattern.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/microsoft/LLMLingua">LLMLingua</a></b></td><td align="center">6.5k</td><td>Prompt-compression research (LLMLingua / LLMLingua-2) — inspired our async, code-safe, fail-open <code>llmlingua</code> engine.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/atjsh/llmlingua-2-js">llmlingua-2-js</a></b></td><td align="center">30</td><td>The JS/ONNX port (MobileBERT / XLM-RoBERTa) used as the worker-thread backend for our LLMLingua engine.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/leninejunior/troglodita">Troglodita</a></b></td><td align="center">26</td><td>PT-BR token compression — powers our pt-BR language pack: pleonasm reduction and filler removal tuned for Brazilian-Portuguese grammar.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/DietrichGebert/ponytail">ponytail</a></b></td><td align="center">86.0k</td><td>The viral "lazy senior dev" YAGNI-coder skill — inspired our <b>less-code</b> Output Style: smallest-working-change steering that cuts _generated_ code (the output-axis sibling to Caveman's terse prose).</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/JuliusBrussee/caveman">Caveman</a></b></td><td align="center">100,538</td><td>The viral "why use many token when few token do trick" project — its caveman-speak philosophy powers our standard compression mode and 30+ filler/condensation rules.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/rtk-ai/rtk">RTK – Rust Token Killer</a></b></td><td align="center">77,185</td><td>High-performance command-output compression — inspired our RTK engine, JSON filter DSL, raw-output recovery and the stacked RTK → Caveman pipeline.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/headroomlabs-ai/headroom">headroom</a></b></td><td align="center">67,310</td><td>Reversible context-compression (SmartCrusher) — inspired our <code>headroom</code> engine and the <code>ccr</code> retrieve-marker pattern.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/microsoft/LLMLingua">LLMLingua</a></b></td><td align="center">6,598</td><td>Prompt-compression research (LLMLingua / LLMLingua-2) — inspired our async, code-safe, fail-open <code>llmlingua</code> engine.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/atjsh/llmlingua-2-js">llmlingua-2-js</a></b></td><td align="center">31</td><td>The JS/ONNX port (MobileBERT / XLM-RoBERTa) used as the worker-thread backend for our LLMLingua engine.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/leninejunior/troglodita">Troglodita</a></b></td><td align="center">40</td><td>PT-BR token compression — powers our pt-BR language pack: pleonasm reduction and filler removal tuned for Brazilian-Portuguese grammar.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/DietrichGebert/ponytail">ponytail</a></b></td><td align="center">108,957</td><td>The viral "lazy senior dev" YAGNI-coder skill — inspired our <b>less-code</b> Output Style: smallest-working-change steering that cuts _generated_ code (the output-axis sibling to Caveman's terse prose).</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ayghri/i-have-adhd">i-have-adhd</a></b></td><td align="center">23,526</td><td>Its action-first, ADHD-friendly response style was adapted into OmniRoute's concise output style across five languages.</td></tr>
|
||||
</table>
|
||||
|
||||
### 🧩 Compact formats, token research & code-aware tooling
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/toon-format/toon">TOON</a></b></td><td align="center">24.9k</td><td>Token-Oriented Object Notation — its columnar, header-plus-rows model shaped our tabular compaction stage.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/blackwell-systems/gcf">GCF – Graph Compact Format</a></b></td><td align="center">22</td><td>First inspired our tabular compaction stage; now its zero-dependency, lossless generic-profile encoder is <b>vendored directly</b> as the Headroom codec (MIT, SPDX-marked), with later numeric-domain and count-mismatch correctness fixes.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ooples/token-optimizer-mcp">token-optimizer-mcp</a></b></td><td align="center">444</td><td>Brotli/SQLite cache + per-session context-delta — inspired our <code>session-dedup</code> engine.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/Mibayy/token-savior">token-savior</a></b></td><td align="center">1.1k</td><td>Bash-output compaction + MCP profiles — inspired our compression bail-out discipline and MCP tool-manifest reduction.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ppgranger/token-saver">token-saver</a></b></td><td align="center">117</td><td>Content-aware, per-file-type output compression with failure-aware bail-out — validated our per-type dispatch and minimum-gain skip.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/alexgreensh/token-optimizer">token-optimizer</a></b></td><td align="center">1.7k</td><td>"Find the ghost tokens" — its offload + recoverable-handle pattern informed our CCR offload thinking.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/Shweta-Mishra-ai/tokenmizer">TokenMizer</a></b></td><td align="center">16</td><td>A session-graph + cross-turn line-dedup blueprint that informed our session-dedup design.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/toon-format/toon">TOON</a></b></td><td align="center">25,233</td><td>Token-Oriented Object Notation — its columnar, header-plus-rows model shaped our tabular compaction stage.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/blackwell-systems/gcf">GCF – Graph Compact Format</a></b></td><td align="center">41</td><td>Its compact graph format and generic-profile design informed OmniRoute's tabular compaction and Headroom codec format.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/blackwell-systems/gcf-typescript">gcf-typescript</a></b></td><td align="center">4</td><td>The MIT TypeScript implementation directly vendored and extended as the Headroom generic-profile codec.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ooples/token-optimizer-mcp">token-optimizer-mcp</a></b></td><td align="center">494</td><td>Brotli/SQLite cache + per-session context-delta — inspired our <code>session-dedup</code> engine.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/Mibayy/token-savior">token-savior</a></b></td><td align="center">1,122</td><td>Bash-output compaction + MCP profiles — inspired our compression bail-out discipline and MCP tool-manifest reduction.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/ppgranger/token-saver">token-saver</a></b></td><td align="center">138</td><td>Content-aware, per-file-type output compression with failure-aware bail-out — validated our per-type dispatch and minimum-gain skip.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/alexgreensh/token-optimizer">token-optimizer</a></b></td><td align="center">1,951</td><td>"Find the ghost tokens" — its offload + recoverable-handle pattern informed our CCR offload thinking.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/Shweta-Mishra-ai/tokenmizer">TokenMizer</a></b></td><td align="center">28</td><td>A session-graph + cross-turn line-dedup blueprint that informed our session-dedup design.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/jessefreitas/OmniCompress">OmniCompress</a></b></td><td align="center">3</td><td>Rust columnar-JSON + content-addressed retrieve + cross-message dedup — validated our <code>headroom</code>/<code>ccr</code>/<code>session-dedup</code> engine design and the cache-stable "compressed form is position-independent" invariant.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/atlassian-labs/mcp-compressor">mcp-compressor</a></b></td><td align="center">98</td><td>MCP tool-schema/description compression — informed our MCP tool-manifest cardinality reduction.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/pdavis68/RepoMapper">RepoMapper</a></b></td><td align="center">187</td><td>Aider-style repo-map ranking — informed our repo-map / retrieval-ranking exploration.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/atlassian-labs/mcp-compressor">mcp-compressor</a></b></td><td align="center">113</td><td>MCP tool-schema/description compression — informed our MCP tool-manifest cardinality reduction.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/pdavis68/RepoMapper">RepoMapper</a></b></td><td align="center">197</td><td>Aider-style repo-map ranking — informed our repo-map / retrieval-ranking exploration.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/mrsimpson/quiet-shell-mcp">quiet-shell-mcp</a></b></td><td align="center">4</td><td>Declarative shell-output reduction over MCP — validated our declarative bash-output compaction.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/dsherret/ts-morph">ts-morph</a></b></td><td align="center">6.1k</td><td>TypeScript Compiler API toolkit — inspired our parser-based comment removal that preserves string, template and regex literals.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/dsherret/ts-morph">ts-morph</a></b></td><td align="center">6,162</td><td>TypeScript Compiler API toolkit — inspired our parser-based comment removal that preserves string, template and regex literals.</td></tr>
|
||||
</table>
|
||||
|
||||
### 🧠 Memory & RAG
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/mem0ai/mem0">Mem0</a></b></td><td align="center">61.2k</td><td>Universal memory layer — its proxy-as-write/read-boundary model shaped our memory architecture.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/letta-ai/letta">Letta (MemGPT)</a></b></td><td align="center">23.9k</td><td>Stateful agents with tiered memory — inspired our Context Control & Recovery (CCR) tiered model.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/onestardao/WFGY">WFGY</a></b></td><td align="center">1.8k</td><td>The ProblemMap taxonomy of 16 recurring RAG/LLM failure modes — the shared vocabulary in our troubleshooting guide.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/mem0ai/mem0">Mem0</a></b></td><td align="center">63,902</td><td>Universal memory layer — its proxy-as-write/read-boundary model shaped our memory architecture.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/letta-ai/letta">Letta (MemGPT)</a></b></td><td align="center">24,382</td><td>Stateful agents with tiered memory — inspired our Context Control & Recovery (CCR) tiered model.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/onestardao/WFGY">WFGY</a></b></td><td align="center">1,781</td><td>The ProblemMap taxonomy of 16 recurring RAG/LLM failure modes — the shared vocabulary in our troubleshooting guide.</td></tr>
|
||||
</table>
|
||||
|
||||
### 🛰️ Traffic inspection, MITM & transparent proxy
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/chouzz/llm-interceptor">llm-interceptor</a></b></td><td align="center">49</td><td>MITM interception/analysis of coding-assistant ↔ LLM traffic — our Traffic Inspector ports its SSE merge, conversation normalization, host passthrough and secret masking (MIT).</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/InterceptSuite/ProxyBridge">ProxyBridge</a></b></td><td align="center">5.5k</td><td>Transparent per-process proxy routing — inspired our crash-safe MITM teardown, socket idle-timeouts, <code>/proc</code> process attribution and TPROXY capture.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/chouzz/llm-interceptor">llm-interceptor</a></b></td><td align="center">66</td><td>MITM interception/analysis of coding-assistant ↔ LLM traffic — our Traffic Inspector ports its SSE merge, conversation normalization, host passthrough and secret masking. The upstream's complete license text is still under provenance review.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/InterceptSuite/ProxyBridge">ProxyBridge</a></b></td><td align="center">5,995</td><td>Transparent per-process proxy routing — inspired our crash-safe MITM teardown, socket idle-timeouts, <code>/proc</code> process attribution and TPROXY capture.</td></tr>
|
||||
</table>
|
||||
|
||||
### 📚 Model data, observability & UI
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/anomalyco/models.dev">models.dev</a></b></td><td align="center">6.0k</td><td>Open database of AI model specs, pricing and capabilities — synced natively into our model catalog.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/xyflow/xyflow">React Flow / xyflow</a></b></td><td align="center">37.7k</td><td>The node-based graph library powering our real-time Compression Studio and Combo/Routing Studio.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/langchain-ai/langgraph">LangGraph</a></b></td><td align="center">37.6k</td><td>LangGraph Studio's live workflow-graph visualization inspired our Studios' real-time cascade view.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/langfuse/langfuse">Langfuse</a></b></td><td align="center">31.4k</td><td>Its trace → span → generation observability model shaped our Compression Studio waterfall.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/kiali/kiali">Kiali</a></b></td><td align="center">3.6k</td><td>Istio service-mesh observability — inspired our circuit-breaker badges and error-edge visuals in the Routing/Combo Studio.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/lobehub/lobe-icons">lobe-icons</a></b></td><td align="center">2.2k</td><td>AI/LLM brand logos that render the provider icons across our dashboard.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/anomalyco/models.dev">models.dev</a></b></td><td align="center">6,555</td><td>Open database of AI model specs, pricing and capabilities — synced natively into our model catalog.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/xyflow/xyflow">React Flow / xyflow</a></b></td><td align="center">38,108</td><td>The node-based graph library powering our real-time Compression Studio and Combo/Routing Studio.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/langchain-ai/langgraph">LangGraph</a></b></td><td align="center">40,314</td><td>LangGraph Studio's live workflow-graph visualization inspired our Studios' real-time cascade view.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/langfuse/langfuse">Langfuse</a></b></td><td align="center">33,592</td><td>Its trace → span → generation observability model shaped our Compression Studio waterfall.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/kiali/kiali">Kiali</a></b></td><td align="center">3,631</td><td>Istio service-mesh observability — inspired our circuit-breaker badges and error-edge visuals in the Routing/Combo Studio.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/lobehub/lobe-icons">lobe-icons</a></b></td><td align="center">2,428</td><td>AI/LLM brand logos that render the provider icons across our dashboard.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/lipis/flag-icons">flag-icons</a></b></td><td align="center">12,354</td><td>Provides the MIT-licensed SVG flags used by the README language selector.</td></tr>
|
||||
</table>
|
||||
|
||||
### 🛡️ Security
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/tldrsec/awesome-secure-defaults">awesome-secure-defaults</a></b></td><td align="center">710</td><td>A curated list of secure-by-default libraries that guides our security choices (Helmet.js, DOMPurify, ssrf-req-filter, safe-regex, Google Tink).</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/tldrsec/awesome-secure-defaults">awesome-secure-defaults</a></b></td><td align="center">721</td><td>A curated list of secure-by-default libraries that guides our security choices (Helmet.js, DOMPurify, ssrf-req-filter, safe-regex, Google Tink).</td></tr>
|
||||
</table>
|
||||
|
||||
### 🧭 Complementary tools
|
||||
|
||||
<table>
|
||||
<tr><th align="left">Project</th><th align="center">⭐</th><th align="left">How it inspired OmniRoute</th></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/BlockRunAI/ClawRouter">ClawRouter</a></b></td><td align="center">6,564</td><td>Inspired request deduplication, emergency zero-cost fallback, pluggable Auto-Combo strategies and multilingual intent classification.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/lbjlaq/Antigravity-Manager">Antigravity-Manager</a></b></td><td align="center">30,652</td><td>Its account-aware model remapping, executable-path validation and plan-label behavior informed OmniRoute's Antigravity runtime.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/jlcodes99/vscode-antigravity-cockpit">vscode-antigravity-cockpit</a></b></td><td align="center">4,817</td><td>Its compact quota-reset countdown format inspired the corresponding provider-limit display in OmniRoute.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/iOfficeAI/AionUi">AionUi</a></b></td><td align="center">32,230</td><td>Its ACP integrations inspired OmniRoute's automatic detection of installed CLI agents.</td></tr>
|
||||
<tr><td nowrap><b><a href="https://github.com/steipete/CodexBar">CodexBar</a></b></td><td align="center">20,507</td><td>Identified the Grok Build quota surface; OmniRoute then verified and corrected the live wire format independently.</td></tr>
|
||||
</table>
|
||||
|
||||
## 📄 License
|
||||
@@ -1589,7 +1666,7 @@ MIT License - see [LICENSE](LICENSE) for details.
|
||||
|
||||
**[⬆ Back to top](#-omniroute)** · Built with ❤️ for the open-source AI community.
|
||||
|
||||
<sub>OmniRoute v3.8.49 · Node ≥22.22.2 · MIT License · <a href="https://omniroute.online">omniroute.online</a></sub>
|
||||
<sub>OmniRoute v3.8.50 · Node ≥22.22.2 · MIT License · <a href="https://omniroute.online">omniroute.online</a></sub>
|
||||
|
||||
</div>
|
||||
<!-- GitHub Discussions enabled for community Q&A -->
|
||||
|
||||
@@ -169,7 +169,8 @@ export async function runSetupClaudeCommand(opts = {}) {
|
||||
let detail = `HTTP ${res.status}`;
|
||||
try {
|
||||
const errorBody = await res.json();
|
||||
const serverMsg = errorBody?.error?.message || errorBody?.error || errorBody?.message || "";
|
||||
const serverMsg =
|
||||
errorBody?.error?.message || errorBody?.error || errorBody?.message || "";
|
||||
if (serverMsg) detail += ` — ${serverMsg}`;
|
||||
} catch {}
|
||||
throw new Error(detail);
|
||||
|
||||
@@ -5,6 +5,7 @@ import { fileURLToPath } from "node:url";
|
||||
import { execFile } from "node:child_process";
|
||||
import { promisify } from "node:util";
|
||||
import { t } from "../i18n.mjs";
|
||||
import { npmBin, npmExecOptions } from "../npm-exec.mjs";
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
@@ -31,9 +32,13 @@ export async function getCurrentVersion() {
|
||||
// they were already on the latest version (#4376). `execFn` is injectable for tests.
|
||||
export async function getLatestVersion(execFn = execFileAsync) {
|
||||
try {
|
||||
const { stdout } = await execFn("npm", ["view", "omniroute", "version", "--prefer-online"], {
|
||||
timeout: 15000,
|
||||
});
|
||||
// argv is all literals, so enabling the shell on win32 cannot splice a
|
||||
// runtime value into the command line (Hard Rule #13).
|
||||
const { stdout } = await execFn(
|
||||
npmBin(),
|
||||
["view", "omniroute", "version", "--prefer-online"],
|
||||
npmExecOptions(process.platform, { timeoutMs: 15000 })
|
||||
);
|
||||
return stdout.trim();
|
||||
} catch {
|
||||
return null;
|
||||
@@ -114,9 +119,11 @@ export async function runUpdateCommand(opts = {}) {
|
||||
|
||||
if (showChangelog) {
|
||||
try {
|
||||
const { stdout } = await execFileAsync("npm", ["view", "omniroute", "changelog"], {
|
||||
timeout: 10000,
|
||||
});
|
||||
const { stdout } = await execFileAsync(
|
||||
npmBin(),
|
||||
["view", "omniroute", "changelog"],
|
||||
npmExecOptions(process.platform, { timeoutMs: 15000 })
|
||||
);
|
||||
if (stdout.trim()) {
|
||||
console.log(stdout.trim());
|
||||
} else {
|
||||
|
||||
@@ -26,7 +26,8 @@
|
||||
"testFailed": "Teste do provedor falhou: {error}",
|
||||
"loginEnabled": "Login: habilitado (senha atualizada)",
|
||||
"loginDisabled": "Login: desabilitado",
|
||||
"providerInfo": "Provedor: {info}"
|
||||
"providerInfo": "Provedor: {info}",
|
||||
"opencode": "Instala e configura o plugin @omniroute/opencode-plugin incluído para o OpenCode"
|
||||
},
|
||||
"doctor": {
|
||||
"title": "OmniRoute Doctor",
|
||||
@@ -254,7 +255,9 @@
|
||||
"no_recovery": "Desabilitar reinício automático em crash (modo debug)",
|
||||
"max_restarts": "Máximo de reinícios em 30s antes de desistir (padrão: 2)",
|
||||
"tray": "Mostrar ícone na bandeja do sistema (apenas desktop, opt-in)",
|
||||
"no_tray": "Desabilitar ícone na bandeja do sistema"
|
||||
"no_tray": "Desabilitar ícone na bandeja do sistema",
|
||||
"tls_cert": "Caminho para um certificado TLS (PEM) para servir HTTPS (também OMNIROUTE_TLS_CERT)",
|
||||
"tls_key": "Caminho para a chave privada TLS (PEM) para servir HTTPS (também OMNIROUTE_TLS_KEY)"
|
||||
},
|
||||
"backup": {
|
||||
"title": "Backup",
|
||||
|
||||
@@ -38,7 +38,8 @@
|
||||
"testFailed": "提供者测试失败:{error}",
|
||||
"loginEnabled": "登录:已启用(密码已更新)",
|
||||
"loginDisabled": "登录:已禁用",
|
||||
"providerInfo": "提供者:{info}"
|
||||
"providerInfo": "提供者:{info}",
|
||||
"opencode": "安装并配置随附的 @omniroute/opencode-plugin 以用于 OpenCode"
|
||||
},
|
||||
"doctor": {
|
||||
"title": "OmniRoute 诊断",
|
||||
@@ -252,7 +253,9 @@
|
||||
"no_recovery": "禁用崩溃自动重启(调试模式)",
|
||||
"max_restarts": "30 秒内的最大崩溃重启次数(默认:2)",
|
||||
"tray": "显示系统托盘图标(仅桌面,选择加入)",
|
||||
"no_tray": "禁用系统托盘图标"
|
||||
"no_tray": "禁用系统托盘图标",
|
||||
"tls_cert": "用于提供 HTTPS 服务的 TLS 证书(PEM)路径(也可用 OMNIROUTE_TLS_CERT)",
|
||||
"tls_key": "用于提供 HTTPS 服务的 TLS 私钥(PEM)路径(也可用 OMNIROUTE_TLS_KEY)"
|
||||
},
|
||||
"backup": {
|
||||
"title": "备份",
|
||||
@@ -1258,5 +1261,69 @@
|
||||
"search": "搜索 npm 注册表中的可用插件",
|
||||
"update": "更新已安装的插件",
|
||||
"scaffold": "搭建新的插件模板"
|
||||
},
|
||||
"authExport": {
|
||||
"description": "导出已解密的提供者凭据(仅限本地,明文输出)",
|
||||
"idOpt": "仅导出与此 id/名称/提供者匹配的连接",
|
||||
"formatOpt": "输出格式:json 或 env",
|
||||
"outOpt": "将输出写入文件而非标准输出(以 0600 权限写入)",
|
||||
"forceOpt": "确认你了解此操作会打印/写入明文密钥",
|
||||
"warning": "⚠ 此操作会打印/写入已解密的明文 API 密钥和 OAuth 令牌。请确保你的屏幕、shell 历史记录以及任何输出文件保持私密。",
|
||||
"confirmHeading": "⚠ 警告:此操作会以明文导出已解密的提供者凭据",
|
||||
"confirmBody": "此命令会为所选连接解密并打印/写入 apiKey、accessToken、refreshToken 和\nidToken。请将输出视为机密。",
|
||||
"confirmFooter": "如需确认,请运行:\n omniroute auth export --force",
|
||||
"missingKey": "导出凭据需要 STORAGE_ENCRYPTION_KEY。",
|
||||
"notFound": "未找到连接:{id}",
|
||||
"invalidFormat": "无效格式:{format}。请使用 json 或 env。"
|
||||
},
|
||||
"radar": {
|
||||
"description": "检查并同步本地 Radar 目录订阅源",
|
||||
"status": "显示本地 Radar 设置和订阅源缓存状态",
|
||||
"sync": "通过本地服务器同步目录、推荐、优惠和 Intel"
|
||||
},
|
||||
"launch": {
|
||||
"description": "启动指向 OmniRoute 的 Claude Code(本地或远程,使用 --profile)",
|
||||
"token": "Claude 客户端应发送的令牌(ANTHROPIC_AUTH_TOKEN)",
|
||||
"notRunning": "无法在 {port} 访问 OmniRoute。请使用 “omniroute serve” 启动它。",
|
||||
"notFound": "在 PATH 中未找到 “claude” CLI。"
|
||||
},
|
||||
"run": {
|
||||
"description": "通过 OmniRoute 启动受支持的 CLI 目标"
|
||||
},
|
||||
"setupClaude": {
|
||||
"description": "从 OmniRoute 模型目录生成 ~/.claude/profiles 的 Claude Code 配置文件"
|
||||
},
|
||||
"connect": {
|
||||
"description": "连接到远程 OmniRoute 服务器并进入远程模式"
|
||||
},
|
||||
"tokens": {
|
||||
"description": "管理限定范围的 CLI 访问令牌(远程模式)"
|
||||
},
|
||||
"configure": {
|
||||
"description": "从活动服务器选择提供者+模型并配置受支持的本地 CLI"
|
||||
},
|
||||
"launchCodex": {
|
||||
"description": "启动指向 OmniRoute 的 Codex CLI(本地或远程 VPS)"
|
||||
},
|
||||
"setupCodex": {
|
||||
"description": "从 OmniRoute 实时模型目录生成 ~/.codex 配置文件"
|
||||
},
|
||||
"packs": {
|
||||
"description": "管理可选的运行时包(ML / 浏览器自动化)",
|
||||
"listDescription": "列出可选包及其安装状态",
|
||||
"installDescription": "将可选包安装到 DATA_DIR",
|
||||
"verifyDescription": "根据随附的校验和索引验证已安装的包",
|
||||
"removeDescription": "移除已安装的可选包",
|
||||
"sourceOpt": "存放包负载和包索引的目录",
|
||||
"warnNoIndex": "未找到 optional-packs.index.json —— 此检出无法进行安装/验证(桌面捆绑包会附带它)",
|
||||
"errUnknown": "未知的包:{name}",
|
||||
"errNoIndex": "未找到包索引;请通过 --source <dir> 传入存放包负载的目录(桌面捆绑包会将其附带在应用旁)",
|
||||
"installed": "包 “{name}” 已安装并在 {dir} 验证通过",
|
||||
"restartHint": "请重启 OmniRoute 服务器(或桌面应用),以便运行时加载该包",
|
||||
"removed": "包 “{name}” 已移除",
|
||||
"notInstalled": "包 “{name}” 未安装",
|
||||
"verifyOk": "所有已安装的包均已验证通过",
|
||||
"verifyFailed": "{count} 个包验证失败",
|
||||
"noneInstalled": "未安装可选包"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -38,7 +38,8 @@
|
||||
"testFailed": "提供者測試失敗:{error}",
|
||||
"loginEnabled": "登入:已啟用(密碼已更新)",
|
||||
"loginDisabled": "登入:已停用",
|
||||
"providerInfo": "提供者:{info}"
|
||||
"providerInfo": "提供者:{info}",
|
||||
"opencode": "安裝並配置隨附的 @omniroute/opencode-plugin 以用於 OpenCode"
|
||||
},
|
||||
"doctor": {
|
||||
"title": "OmniRoute 診斷",
|
||||
@@ -252,7 +253,9 @@
|
||||
"no_recovery": "停用崩潰自動重啟(除錯模式)",
|
||||
"max_restarts": "30 秒內的最大崩潰重啟次數(預設:2)",
|
||||
"tray": "顯示系統托盤圖示(僅桌面,選擇加入)",
|
||||
"no_tray": "停用系統托盤圖示"
|
||||
"no_tray": "停用系統托盤圖示",
|
||||
"tls_cert": "用於提供 HTTPS 服務的 TLS 憑證(PEM)路徑(也可用 OMNIROUTE_TLS_CERT)",
|
||||
"tls_key": "用於提供 HTTPS 服務的 TLS 私鑰(PEM)路徑(也可用 OMNIROUTE_TLS_KEY)"
|
||||
},
|
||||
"backup": {
|
||||
"title": "備份",
|
||||
@@ -1258,5 +1261,69 @@
|
||||
"search": "搜尋 npm 登錄檔中的可用外掛",
|
||||
"update": "更新已安裝的外掛",
|
||||
"scaffold": "搭建新的外掛模板"
|
||||
},
|
||||
"authExport": {
|
||||
"description": "匯出已解密的提供者憑據(僅限本機,明文輸出)",
|
||||
"idOpt": "僅匯出與此 id/名稱/提供者相符的連線",
|
||||
"formatOpt": "輸出格式:json 或 env",
|
||||
"outOpt": "將輸出寫入檔案而非標準輸出(以 0600 權限寫入)",
|
||||
"forceOpt": "確認你了解此操作會列印/寫入明文密鑰",
|
||||
"warning": "⚠ 此操作會列印/寫入已解密的明文 API 金鑰和 OAuth 令牌。請確保你的螢幕、shell 歷史記錄以及任何輸出檔案保持私密。",
|
||||
"confirmHeading": "⚠ 警告:此操作會以明文匯出已解密的提供者憑據",
|
||||
"confirmBody": "此命令會為所選連線解密並列印/寫入 apiKey、accessToken、refreshToken 和\nidToken。請將輸出視為機密。",
|
||||
"confirmFooter": "如需確認,請執行:\n omniroute auth export --force",
|
||||
"missingKey": "匯出憑據需要 STORAGE_ENCRYPTION_KEY。",
|
||||
"notFound": "找不到連線:{id}",
|
||||
"invalidFormat": "無效格式:{format}。請使用 json 或 env。"
|
||||
},
|
||||
"radar": {
|
||||
"description": "檢查並同步本機 Radar 目錄訂閱來源",
|
||||
"status": "顯示本機 Radar 設定和訂閱來源快取狀態",
|
||||
"sync": "透過本機伺服器同步目錄、推薦、優惠和 Intel"
|
||||
},
|
||||
"launch": {
|
||||
"description": "啟動指向 OmniRoute 的 Claude Code(本機或遠端,使用 --profile)",
|
||||
"token": "Claude 用戶端應傳送的令牌(ANTHROPIC_AUTH_TOKEN)",
|
||||
"notRunning": "無法在 {port} 存取 OmniRoute。請使用「omniroute serve」啟動它。",
|
||||
"notFound": "在 PATH 中找不到「claude」CLI。"
|
||||
},
|
||||
"run": {
|
||||
"description": "透過 OmniRoute 啟動受支援的 CLI 目標"
|
||||
},
|
||||
"setupClaude": {
|
||||
"description": "從 OmniRoute 模型目錄產生 ~/.claude/profiles 的 Claude Code 配置檔"
|
||||
},
|
||||
"connect": {
|
||||
"description": "連線到遠端 OmniRoute 伺服器並進入遠端模式"
|
||||
},
|
||||
"tokens": {
|
||||
"description": "管理限定範圍的 CLI 存取令牌(遠端模式)"
|
||||
},
|
||||
"configure": {
|
||||
"description": "從使用中的伺服器選擇提供者+模型並配置受支援的本機 CLI"
|
||||
},
|
||||
"launchCodex": {
|
||||
"description": "啟動指向 OmniRoute 的 Codex CLI(本機或遠端 VPS)"
|
||||
},
|
||||
"setupCodex": {
|
||||
"description": "從 OmniRoute 即時模型目錄產生 ~/.codex 配置檔"
|
||||
},
|
||||
"packs": {
|
||||
"description": "管理可選的執行階段套件(ML / 瀏覽器自動化)",
|
||||
"listDescription": "列出可選套件及其安裝狀態",
|
||||
"installDescription": "將可選套件安裝到 DATA_DIR",
|
||||
"verifyDescription": "根據隨附的總和檢查碼索引驗證已安裝的套件",
|
||||
"removeDescription": "移除已安裝的可選套件",
|
||||
"sourceOpt": "存放套件負載和套件索引的目錄",
|
||||
"warnNoIndex": "找不到 optional-packs.index.json —— 此檢出無法進行安裝/驗證(桌面套件會隨附它)",
|
||||
"errUnknown": "未知的套件:{name}",
|
||||
"errNoIndex": "找不到套件索引;請透過 --source <dir> 傳入存放套件負載的目錄(桌面套件會將其隨附在應用程式旁)",
|
||||
"installed": "套件「{name}」已安裝並在 {dir} 驗證通過",
|
||||
"restartHint": "請重新啟動 OmniRoute 伺服器(或桌面應用程式),以便執行階段載入該套件",
|
||||
"removed": "套件「{name}」已移除",
|
||||
"notInstalled": "套件「{name}」未安裝",
|
||||
"verifyOk": "所有已安裝的套件均已驗證通過",
|
||||
"verifyFailed": "{count} 個套件驗證失敗",
|
||||
"noneInstalled": "未安裝可選套件"
|
||||
}
|
||||
}
|
||||
|
||||
34
bin/cli/npm-exec.mjs
Normal file
@@ -0,0 +1,34 @@
|
||||
// Spawning npm from the CLI, on every platform.
|
||||
//
|
||||
// On Windows npm is `npm.cmd`, a batch wrapper. Node ≥ 24 refuses to spawn a
|
||||
// `.cmd` without a shell (nodejs/node#52554), and a bare `npm` can additionally
|
||||
// resolve to an extensionless shim that `CreateProcess` cannot execute — so the
|
||||
// call fails with `EINVAL` or `ENOENT` while npm works fine in the same terminal.
|
||||
// `src/lib/services/installers/utils.ts` already solves this for the server; this
|
||||
// is the same rule for the `bin/cli` entry points, which cannot import TypeScript.
|
||||
//
|
||||
// SECURITY (Hard Rule #13): enabling the shell means the SHELL splits the command
|
||||
// line, not `execFile`. Every argv element passed alongside these options must be
|
||||
// a literal — never a runtime value — or it must be validated first. Callers that
|
||||
// need to pass a user-supplied name have to guard it themselves.
|
||||
|
||||
/** The npm binary to spawn on this platform. */
|
||||
export function npmBin(platform = process.platform) {
|
||||
const isBun = Boolean(process.versions.bun);
|
||||
if (platform === "win32") return isBun ? "bun.exe" : "npm.cmd";
|
||||
return isBun ? "bun" : "npm";
|
||||
}
|
||||
|
||||
/**
|
||||
* `execFile` / `spawnSync` options for an npm call.
|
||||
*
|
||||
* @param {NodeJS.Platform} platform
|
||||
* @param {{ timeoutMs?: number, stdio?: string }} [options]
|
||||
*/
|
||||
export function npmExecOptions(platform = process.platform, options = {}) {
|
||||
const base = {};
|
||||
if (options.timeoutMs !== undefined) base.timeout = options.timeoutMs;
|
||||
if (options.stdio !== undefined) base.stdio = options.stdio;
|
||||
if (platform !== "win32") return { ...base, shell: false };
|
||||
return { ...base, shell: true, windowsHide: true };
|
||||
}
|
||||
@@ -2,6 +2,7 @@ import { existsSync, mkdirSync, writeFileSync, chmodSync } from "node:fs";
|
||||
import { join } from "node:path";
|
||||
import { homedir } from "node:os";
|
||||
import { execSync } from "node:child_process";
|
||||
import { pathToFileURL } from "node:url";
|
||||
|
||||
const RUNTIME_DIR = join(homedir(), ".omniroute", "runtime");
|
||||
// systray2 is a maintained fork with prebuilt binaries — installed lazily at runtime,
|
||||
@@ -16,6 +17,16 @@ export const SYSTRAY_PACKAGE = "systray2";
|
||||
export const SYSTRAY_VERSION = "2.1.4";
|
||||
const SYSTRAY_SPEC = `${SYSTRAY_PACKAGE}@${SYSTRAY_VERSION}`;
|
||||
|
||||
// Dynamic `import()` resolves its specifier as a URL, not a filesystem path.
|
||||
// On Windows the lazily-installed systray2 lives at an absolute path whose
|
||||
// leading drive letter the ESM loader parses as an unsupported URL scheme
|
||||
// (e.g. `c:`) and rejects. Build a file:// URL so the tray import works on
|
||||
// Windows too. Same defect fixed for the CLI db-fallback imports in #11238,
|
||||
// missed at this call site.
|
||||
export function systrayModuleSpecifier(runtimeDir: string): string {
|
||||
return pathToFileURL(join(runtimeDir, "node_modules", SYSTRAY_PACKAGE)).href;
|
||||
}
|
||||
|
||||
export function resolveSystrayBinName(platform: NodeJS.Platform): string | null {
|
||||
if (platform === "win32") return "tray_windows_release.exe";
|
||||
if (platform === "darwin") return "tray_darwin_release";
|
||||
@@ -60,8 +71,7 @@ export async function loadSystray(): Promise<(new (...args: unknown[]) => unknow
|
||||
// drop the +x bit on extraction (observed on macOS).
|
||||
chmodSystrayBinAt(RUNTIME_DIR, process.platform);
|
||||
try {
|
||||
const modPath = join(RUNTIME_DIR, "node_modules", SYSTRAY_PACKAGE);
|
||||
const mod = await import(modPath);
|
||||
const mod = await import(systrayModuleSpecifier(RUNTIME_DIR));
|
||||
return (mod.default ?? mod.SysTray ?? mod) as (new (...args: unknown[]) => unknown) | null;
|
||||
} catch (err) {
|
||||
console.warn(`[omniroute] tray runtime import failed: ${(err as Error).message}`);
|
||||
|
||||
@@ -114,10 +114,13 @@ function writeLinuxSystemdUnit(cliPath) {
|
||||
const unitDir = dirname(linuxSystemdUnitPath());
|
||||
mkdirSync(unitDir, { recursive: true });
|
||||
const envFile = join(userHomeDir(), ".omniroute", ".env");
|
||||
const nodeBinDir = dirname(process.execPath);
|
||||
const userLocalBin = join(userHomeDir(), ".local", "bin");
|
||||
const pathEnv = `${nodeBinDir}:${userLocalBin}:/usr/local/sbin:/usr/local/bin:/usr/bin:/bin`;
|
||||
const lines = [
|
||||
"[Unit]",
|
||||
"Description=OmniRoute AI proxy router",
|
||||
"After=network-online.target",
|
||||
"After=network-online.target graphical-session.target",
|
||||
"Wants=network-online.target",
|
||||
"",
|
||||
"[Service]",
|
||||
@@ -134,6 +137,7 @@ function writeLinuxSystemdUnit(cliPath) {
|
||||
`ExecStart=${buildServeExecLine(cliPath, { tray: false })}`,
|
||||
"Restart=on-failure",
|
||||
"RestartSec=5",
|
||||
`Environment="PATH=${pathEnv}"`,
|
||||
];
|
||||
if (existsSync(envFile)) lines.push(`EnvironmentFile=-${envFile}`);
|
||||
lines.push("", "[Install]", "WantedBy=default.target", "");
|
||||
|
||||
1
changelog.d/features/10556-elevenlabs-native-routes.md
Normal file
@@ -0,0 +1 @@
|
||||
- **feat(audio):** proxy native ElevenLabs voices, text-to-speech, and speech-to-text HTTP routes through stored OmniRoute credentials, preserving query strings, multipart uploads, binary responses, and upstream errors (#10556).
|
||||
1
changelog.d/features/10590-google-ai-studio-tts.md
Normal file
@@ -0,0 +1 @@
|
||||
- Added Google AI Studio Gemini batch text-to-speech support through `POST /v1/audio/speech`.
|
||||
3
changelog.d/features/11023-compression-worker-pool.md
Normal file
@@ -0,0 +1,3 @@
|
||||
- Run synchronous RTK and Caveman request compression in a bounded worker-thread pool, keeping
|
||||
large `/v1/responses` compression heaps outside the HTTP isolate while preserving strict
|
||||
fail-open behavior and per-engine telemetry.
|
||||
@@ -0,0 +1 @@
|
||||
- **feat(video bridge):** harden the optional drill-down cache substrate with exact-path broker policy, canonical principal/session/media isolation, independent retained-byte quotas, cancellation-safe commits, rejection of excess or non-canonical Base64 padding and non-JPEG/truncated media, warning-sensitive full JPEG canonicalization that strips trailing polyglot bytes, server-derived dimensions, and auditable derivation metadata; production tenant binding and multi-resolution selection remain follow-up work ([#11369](https://github.com/diegosouzapw/OmniRoute/pull/11369))
|
||||
1
changelog.d/features/11383-video-bridge-focused-mode.md
Normal file
@@ -0,0 +1 @@
|
||||
- **feat(video):** add an opt-in focused analysis mode that safely uses a normalized, 500-code-point latest-user hint for task-aware frame captions while preserving full-mode prompts, temporal-window isolation, and cache identity without storing raw task text ([#11383](https://github.com/diegosouzapw/OmniRoute/pull/11383)).
|
||||
1
changelog.d/features/6342-cliproxy-account-health.md
Normal file
@@ -0,0 +1 @@
|
||||
- feat(services): show sanitized CLIProxyAPI account health from its authenticated management API without exposing credentials, file paths, or raw account metadata (#6342)
|
||||
1
changelog.d/fixes/10352-github-access-token-health.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(github):** proactive credential health now verifies GitHub access tokens through the existing Copilot token exchange, marks only a confirmed `401 Unauthorized` as expired, and leaves rate limits, permission failures, upstream failures, and network errors routable ([#10352](https://github.com/diegosouzapw/OmniRoute/issues/10352)) — thanks @RaviTharuma
|
||||
@@ -0,0 +1 @@
|
||||
- **fix(providers):** Antigravity OAuth marks connects with no Cloud Code projectId as degraded instead of a false "Connected"; BYOP detection at connect time, auto-disable of confirmed-missing accounts, and selection-side rotation ([#11284](https://github.com/diegosouzapw/OmniRoute/issues/11284))
|
||||
@@ -0,0 +1 @@
|
||||
- **fix(db):** group model patterns escape regex metacharacters, so `gpt-4.1*` no longer matches `gpt-4o1-preview` and a pattern like `gpt-4(*` no longer throws `SyntaxError` out of the completion and `/v1/models` paths ([#11311](https://github.com/diegosouzapw/OmniRoute/pull/11311))
|
||||
1
changelog.d/fixes/11319-upstream-proxy-host-spelling.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(db):** the upstream proxy URL check judges the host by address instead of by spelling, so `http://[::ffff:169.254.169.254]`, `[::ffff:10.0.0.5]`, ULA/link-local and CGNAT targets are refused like their dotted equivalents ([#11319](https://github.com/diegosouzapw/OmniRoute/pull/11319))
|
||||
1
changelog.d/fixes/11325-i18n-pt-placeholder-parity.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(i18n):** three `pt` strings had dropped their placeholders — the cache tile's subtitle repeated its own label instead of showing `{total}` — and a unit test now enforces placeholder parity with `en` across all locales ([#11325](https://github.com/diegosouzapw/OmniRoute/pull/11325))
|
||||
1
changelog.d/fixes/11326-kie-market-google-imagen-ids.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(kie):** map the remaining `google-imagen/*` KIE Market catalog ids (`nano-banana`, `nano-banana-pro`, `nano-banana-edit`) to their real, KIE-documented upstream `model` values — `#11225`'s fix only covered `nano-banana-2` ([#11326](https://github.com/diegosouzapw/OmniRoute/pull/11326)).
|
||||
1
changelog.d/fixes/11328-upstream-headers-proxy-auth.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(security):** `proxy-authorization` and `proxy-authenticate` are refused as upstream/custom headers, so a proxy credential is no longer forwarded to the model provider — the canonical denylist now matches the RFC 7230 §6.1 set the rest of the codebase already strips ([#11328](https://github.com/diegosouzapw/OmniRoute/pull/11328))
|
||||
@@ -0,0 +1 @@
|
||||
- **fix(video-bridge):** fall back to the deterministic active-window midpoint when a one-frame scene-aware budget cannot preserve both timeline ends; a real FFmpeg fixture matrix now covers rapid cuts, gradual changes, static and short clips, and detector failure ([#11344](https://github.com/diegosouzapw/OmniRoute/pull/11344)).
|
||||
1
changelog.d/fixes/11347-codex-claude-empty-tool-use.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(translator):** Codex Responses tool calls translated for Claude clients no longer emit a duplicate `tool_use` block with the same ID and an empty name, preventing Claude Code from terminating with `No such tool available` ([#11347](https://github.com/diegosouzapw/OmniRoute/pull/11347))
|
||||
@@ -0,0 +1 @@
|
||||
- **fix(video-bridge):** burn high-contrast timestamps into every bounded contact-sheet cell and add a real-model A/B harness whose promotion verdict stays `HOLD` until token, latency, and quality evidence is actually executed ([#11350](https://github.com/diegosouzapw/OmniRoute/pull/11350))
|
||||
1
changelog.d/fixes/11362-video-bridge-result-cache.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(video):** fingerprint protected Video Bridge bytes, coalesce concurrent work, and fail open when the bounded TTL/LRU result cache is unavailable or corrupt ([#11362](https://github.com/diegosouzapw/OmniRoute/pull/11362))
|
||||
1
changelog.d/fixes/11367-catalog-eventloop-9147.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(catalog):** keep large `/v1/models` builds responsive by reusing the build-local capability snapshot throughout enrichment and Auto-Combo preparation, yielding cooperatively while constructing virtual candidate pools, and avoiding unrelated synchronous database diagnostics on the cache-TTL read path ([#11367](https://github.com/diegosouzapw/OmniRoute/pull/11367))
|
||||
1
changelog.d/fixes/11382-video-bridge-dedup-policy.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(video):** apply the caption-frame cap after bounded visual deduplication, preserve first/final candidates plus small high-contrast motion and text changes, and version the dedup policy in result-cache identity ([#11382](https://github.com/diegosouzapw/OmniRoute/pull/11382)).
|
||||
1
changelog.d/fixes/11388-live-ws-handshake-port.md
Normal file
@@ -0,0 +1 @@
|
||||
- **Live dashboard:** honour the WebSocket port reported by `/api/v1/ws?handshake=1` instead of the port compiled into the bundle, so a `LIVE_WS_PORT` override reaches prebuilt Docker/npm images and Combo Studio Live connects behind a reverse proxy ([#11331](https://github.com/diegosouzapw/OmniRoute/issues/11331)).
|
||||
1
changelog.d/fixes/11394-modelsdev-interval-slider.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(dashboard):** Model Database sync interval slider ticks now match the thumb position — checkpoint-space slider with magnetic snap on release ([#11394](https://github.com/diegosouzapw/OmniRoute/pull/11394)) — thanks @An0nym0us92
|
||||
1
changelog.d/fixes/cli-update-npm-win32-11335.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(cli):** `omniroute update` now finds npm on Windows. It called `execFile("npm", …)` with no shell, and on Node ≥ 24 a `.cmd` wrapper cannot be spawned that way (nodejs/node#52554) — while a bare `npm` can also resolve to an extensionless shim `CreateProcess` refuses. The result was `✖ Could not check latest version. Is npm available?` in a terminal where `npm view omniroute version` worked fine, so the updater was unusable on Windows even though nothing was wrong with the install. This is the same class as #5379/#5542, which fixed the server-side calls; the CLI entry points were missed because they are plain `.mjs` and cannot import the TypeScript helper. `bin/cli/npm-exec.mjs` now states the same rule for them: `npm.cmd` plus a shell on win32, no shell anywhere else. Both npm lookups in `update.mjs` (version and changelog) pass a literal argv array, so enabling the shell cannot splice a runtime value into the command line — a test asserts that and fails if a future edit interpolates one. (#11335)
|
||||
1
changelog.d/fixes/compression-worker-bundler-resolve.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(compression):** use `pathToFileURL` in `compressionWorkerPool` so bundlers (Webpack / Turbopack) do not attempt static asset resolution of missing `compressionWorker.js` during build
|
||||
1
changelog.d/fixes/glm-credit-limit-quota.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(usage):** z.ai/GLM coding-plan subscription keys now render their quota cards again, with absolute credits. Z.ai's `/api/monitor/usage/quota/limit` switched these keys from `TOKENS_LIMIT` to `CREDIT_LIMIT` rows (same `unit`/`number` semantics: unit=3/number=5 → 5-hour window, unit=6/number=1 → weekly), and the parser only matched `TOKENS_LIMIT`/`TIME_LIMIT`, so both rows were dropped and the subscription card rendered empty. `CREDIT_LIMIT` is now accepted alongside `TOKENS_LIMIT`, and when the row carries absolute credit fields (`usage`/`currentValue`/`remaining`) they are preferred over the percent-only scale, so the card shows `3341 / 28000` like z.ai's own dashboard instead of `11 / 100`
|
||||
1
changelog.d/fixes/lasterror-provider-error-detail.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(auth):** a connection's `lastError` now names the real upstream failure instead of the bare string `Provider error`. `markAccountUnavailable` kept the reason only when it was already a string, so every other shape collapsed to that literal — and the shape that matters most is not a string: a failed `fetch` arrives as `TypeError: fetch failed` with the actionable part on `error.cause.code`, which means a wrong port, a firewall, a DNS failure and a blocked proxy all looked identical in the dashboard and in the console line. `describeUpstreamFailure` (in `src/shared/utils/upstreamError.ts`, reusing the `extractErrorMessage` that already parsed provider bodies) reads Error messages and appends the transport code when the message does not already carry it, reads the usual provider JSON shapes (`error.message`, `message`, string `error`, `detail`, `errors[]`), and falls back to the code alone before giving up. It never serializes the error object wholesale, so a request body or header attached to an error cannot leak into the stored reason — pinned by a test.
|
||||
1
changelog.d/fixes/live-ws-public-url-runtime.md
Normal file
@@ -0,0 +1 @@
|
||||
- **fix(live-ws):** the Live dashboard socket can now be pointed at a reverse proxy without rebuilding the image. `NEXT_PUBLIC_*` is inlined at BUILD time, so a prebuilt Docker or npm image never carries an operator's `NEXT_PUBLIC_LIVE_WS_PUBLIC_URL` — which is exactly why the browser discovers the socket through `/api/v1/ws?handshake=1` instead. The server side of that handshake, however, read only the `NEXT_PUBLIC_`-prefixed name, so it had nothing to echo: behind Traefik the dashboard kept dialling `wss://<host>:20132/live-ws` and sat on "Live disabled — WebSocket disconnected. Showing last known state." `LIVE_WS_PUBLIC_URL` is now read at runtime alongside the existing `LIVE_WS_HOST` / `LIVE_WS_PORT`, and the prefixed name stays supported as the fallback, so deployments that already set it are unaffected. Only `ws://` and `wss://` values are accepted, matching the guard the client already applies. (#11331)
|
||||
@@ -0,0 +1,3 @@
|
||||
- **fix(deps):** prevent pnpm from auto-installing the unused `@lobehub/ui` peer subtree of
|
||||
`@lobehub/icons`, keeping six unneeded packages with incompatible or unverifiable license
|
||||
metadata out of production installs ([#11342](https://github.com/diegosouzapw/OmniRoute/pull/11342)).
|
||||
@@ -0,0 +1 @@
|
||||
- **ci(changelog):** replace the broad removal bypass with an exact, hash-bound reconciliation ledger and bind merge-train checks to their requested release base ([#11345](https://github.com/diegosouzapw/OmniRoute/pull/11345)).
|
||||
5
changelog.d/maintenance/11356-readme-live-metrics.md
Normal file
@@ -0,0 +1,5 @@
|
||||
- **docs(readme):** reconcile live v3.8.50 provider, free-tier, CLI, routing, test,
|
||||
community, sponsor, acknowledgment, and SVG metrics with their audited source
|
||||
denominators, including a deduplicated OmniRoute-in-Action snapshot and distinct
|
||||
contributor rankings for merged pull requests, GitHub-attributed commits, and Git history
|
||||
([#11356](https://github.com/diegosouzapw/OmniRoute/pull/11356)).
|
||||
@@ -0,0 +1 @@
|
||||
- **docs(openapi):** document the conditionally management-authenticated, same-origin `POST /api/openapi/try` proxy contract and restore the release branch's operation-coverage ratchet ([#11363](https://github.com/diegosouzapw/OmniRoute/pull/11363))
|
||||
@@ -0,0 +1 @@
|
||||
- **fix(video-bridge):** make opt-in segment-aware sampling use one bounded structural FFmpeg pass (scene, freeze, blur, exposure, and SI/TI), preserve long trailing segments, fail open to uniform sampling, and add real-media structural-oracle, overhead, post-dedup caption-call, and false-positive evidence while holding unconfigured model quality and gain-versus-cost claims ([#11381](https://github.com/diegosouzapw/OmniRoute/pull/11381)).
|
||||
@@ -0,0 +1 @@
|
||||
- **test(kimi):** the Kimi background health sweep no longer draws its refresh window inside the assertion. `checkKimiWebConnectionIfNeeded` spreads the refresh over `[60, 240)` seconds before expiry so a fleet of connections does not stampede the token endpoint, and the test used a token expiring in 90 seconds and asserted that a refresh happened — which is true only when the draw lands at 90 or above, i.e. 150 of the 180 possible values. Measured: the test fails 1 run in 6 (16.7% by construction; 4 of 20 local runs), and it is what the Node 26 nightly hit and reported as a Node-compat break (#11361). The spread is now `defaultKimiRefreshJitterSec()` and the window is injectable as `jitterSecFn`, so the test decides it instead of rolling for it; production behaviour is unchanged. Cases were added for a token outside the window and for the default spread's range.
|
||||
@@ -3445,7 +3445,7 @@
|
||||
},
|
||||
"tests/integration/qdrant-routes.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 19
|
||||
"count": 3
|
||||
}
|
||||
},
|
||||
"tests/integration/quota-pools-usage.test.ts": {
|
||||
@@ -4029,10 +4029,10 @@
|
||||
},
|
||||
"tests/unit/cli-combo-suggest-commands.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 16
|
||||
"count": 14
|
||||
},
|
||||
"@typescript-eslint/no-unused-vars": {
|
||||
"count": 2
|
||||
"count": 1
|
||||
}
|
||||
},
|
||||
"tests/unit/cli-completion-dynamic.test.ts": {
|
||||
@@ -4042,7 +4042,7 @@
|
||||
},
|
||||
"tests/unit/cli-compression-commands.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 32
|
||||
"count": 20
|
||||
}
|
||||
},
|
||||
"tests/unit/cli-context-eng-commands.test.ts": {
|
||||
@@ -4099,7 +4099,7 @@
|
||||
},
|
||||
"tests/unit/cli-mcp-call-commands.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 16
|
||||
"count": 10
|
||||
}
|
||||
},
|
||||
"tests/unit/cli-memory-commands.test.ts": {
|
||||
@@ -4130,7 +4130,7 @@
|
||||
},
|
||||
"tests/unit/cli-oneproxy-commands.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 22
|
||||
"count": 14
|
||||
},
|
||||
"@typescript-eslint/no-unused-vars": {
|
||||
"count": 1
|
||||
@@ -4203,9 +4203,6 @@
|
||||
"tests/unit/cli-resilience-commands.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 16
|
||||
},
|
||||
"@typescript-eslint/no-unused-vars": {
|
||||
"count": 2
|
||||
}
|
||||
},
|
||||
"tests/unit/cli-runtime-extended.test.ts": {
|
||||
@@ -4238,7 +4235,7 @@
|
||||
},
|
||||
"tests/unit/cli-skills-commands.test.ts": {
|
||||
"@typescript-eslint/no-explicit-any": {
|
||||
"count": 22
|
||||
"count": 16
|
||||
}
|
||||
},
|
||||
"tests/unit/cli-stop-supervisor-respawn-9455.test.ts": {
|
||||
@@ -6620,4 +6617,4 @@
|
||||
"count": 2
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -227,7 +227,9 @@
|
||||
"tests/unit/translator-resp-gemini-to-openai.test.ts": 1604,
|
||||
"tests/unit/usage-service-hardening.test.ts": 1928,
|
||||
"tests/unit/vscode-token-routes.test.ts": 1633,
|
||||
"tests/unit/executor-antigravity.test.ts": 1427
|
||||
"tests/unit/executor-antigravity.test.ts": 1427,
|
||||
"tests/unit/guardrails/videoBridgeResultCache.test.ts": 1040,
|
||||
"_rebaseline_2026_08_24_video_bridge_fu01_fu03_fu04_result_cache_tests": "PRs #11362 (FU-01 cache hardening) + #11382 (FU-03 visual dedup policy identity) + #11383 (FU-04 focused analysis mode) own test growth: videoBridgeResultCache.test.ts <1000->1040, +40 (sum of three stacked PRs boarded together in the same merge-batch, each adding its own cache-identity assertions on the shared result-cache seam). Owner pre-authorized rebaseline for legitimate PR growth (2026-08-19 directive)."
|
||||
},
|
||||
"_rebaseline_2026_06_09": "Re-baseline consciente pre-release v3.8.19: 9 arquivos cresceram durante o ciclo (features mergeadas: RequestLoggerV2 +281 request-logger rework, stream +101, combo +73, chatCore +45, catalog +32 fable-5/catalog-flag, callLogs +4, accountFallback +2, usageHistory novo 840) + core.ts +7 (fix resetAllDbModuleState, PR 3536). A catraca segue valendo destes valores — proximo crescimento falha. Decisao: encolher (esp. RequestLoggerV2/chatCore) e a issue #3501 ficam para o ciclo seguinte.",
|
||||
"_rebaseline_2026_06_11_phase1f": "Phase 1f (#3501): ProviderDetailPageClient.tsx 4948→4062 (-886 LOC); 3 novos hooks extraídos. useProviderConnections.ts=954 acima do cap=800 — justificado: extração direta do god-component (zero lógica nova), própria redução do cliente supera o custo. useProviderSettings.ts=263 e useProviderModels.ts=154 já abaixo do cap.",
|
||||
@@ -308,7 +310,7 @@
|
||||
"_rebaseline_2026_07_27_v3849_train1h": "Merge-train 1H (31 PRs) — owner-approved 2026-07-27. Two distinct causes, kept separate on purpose: (1) GENUINE irreducible growth at existing chokepoints — providerLimits/auth (#8632 Kimi quota-reset recovery), rateLimitManager (#8616 idle wedged limiters), models-catalog-route.test (#8610 OpenCode Go effort aliases); (2) COLLISION with #8585, which banked shrinks measured on the pre-train release tip while 30 sibling PRs in the SAME train grew those files again — chat/accountFallback (#8628), chatCore (#8613), videoGeneration (#8581), imageGeneration. The zero-headroom frozen entries cannot absorb either. Ceilings re-pinned to the post-merge tip; #8612 (also in this train) automates shrink-banking so this self-inflicted drift stops recurring. Detail: src/lib/usage/providerLimits.ts 1006->1013 (#8632); src/sse/services/auth.ts 2492->2508 (#8632); open-sse/services/rateLimitManager.ts 1014->1060 (#8616); src/sse/handlers/chat.ts 1842->1845 (#8628); open-sse/handlers/chatCore.ts 4939->4955 (#8613); open-sse/handlers/imageGeneration.ts 3100->3101 ((sem PR — teto do #8585)); open-sse/handlers/videoGeneration.ts 1038->1063 (#8581); open-sse/services/accountFallback.ts 1965->1966 (#8628); tests/unit/models-catalog-route.test.ts 1608->1636 (#8610)",
|
||||
"frozen": {
|
||||
"_rebaseline_2026_08_20_10878_10799_provider_health_probes": "PRs #10878 (unsupported OpenAI-like validation probes stay neutral) + #10799 (preserve credential health on inconclusive NVIDIA-timeout/Antigravity-400 probes) own growth: src/app/api/providers/[id]/test/route.ts 946->1025 (+79, sum of both boarded together). Both add narrowly-scoped classification branches at the existing test-route dispatch chokepoint (unsupported-capability skip, credential-inconclusive detection) rather than new files, mirroring the prior 2026_06_27_5193 rebaseline of the same file. Covered by tests/unit/provider-validation-unsupported-neutral.test.ts + tests/unit/provider-health-inconclusive-probes.test.ts.",
|
||||
"src/app/api/providers/[id]/test/route.ts": 1215,
|
||||
"src/app/api/providers/[id]/test/route.ts": 1237,
|
||||
"_rebaseline_2026_08_23_11141_oauth_400_recovery": "PR #11141 (HouMinXi) own growth: test/route.ts 1025->1215 (+190, the reactive-400 recovery path — a fully rebuilt probe for refresh+retry on refreshable non-rotating connections, with inconclusive-status preservation and rotating-provider exclusion; all growth is the new probe builder + guards at the existing test-route dispatch, extraction would split the retry flow mid-logic). Covered by tests/unit/oauth-400-recovery.test.ts (8, bug-injection proof). Owner pre-authorized baseline bumps 2026-08-22.",
|
||||
"_rebaseline_2026_06_22_4644_deepseek_web_tools": "PR #4644 (BugsBag/robust deepseek-web tool-call parsing): open-sse/executors/deepseek-web.ts 1117->1125 (+8). The new agentic tool-call path emits surrounding text + reasoning before tool_calls and swaps to the dedicated deepseekWebTools.ts parser; the +8 lines are cohesive wiring at the existing transformSSE chokepoint (the parser itself lives in the new deepseekWebTools.ts file, already under cap). The PR's own fast-gate (PR->release) does not run check:file-size, so this surfaced only at release reconcile. Covered by tests/unit/deepseek-web-tools-variants.test.ts + deepseek-web-tools-execute.test.ts.",
|
||||
"_rebaseline_2026_06_23_4712_deepseek_web_tool_results": "PR for #4712 (deepseek-web drops role:tool): open-sse/executors/deepseek-web.ts 1125->1148 (+23). messagesToPrompt() now folds role:\"tool\" results into the single-prompt transcript (recovering the tool name from the preceding assistant tool_calls by tool_call_id) instead of silently dropping them; the lines are cohesive wiring inside the existing function. Covered by tests/unit/deepseek-web-tool-result-prompt-4712.test.ts.",
|
||||
@@ -433,7 +435,8 @@
|
||||
"src/shared/components/analytics/charts.tsx": 1346,
|
||||
"src/shared/services/cliRuntime.ts": 1459,
|
||||
"src/sse/handlers/chat.ts": 2493,
|
||||
"src/sse/services/auth.ts": 3344,
|
||||
"src/sse/services/auth.ts": 3346,
|
||||
"_rebaseline_2026_08_24_lasterror_provider_error_detail": "PR (ntdat812) own growth: src/sse/services/auth.ts 3344->3346 (+2). One line is the import of describeUpstreamFailure from @/shared/utils/upstreamError, which replaces the string-only collapse `typeof errorText === \"string\" ? errorText.slice(0, 100) : \"Provider error\"` at the single markAccountUnavailable chokepoint (net 0 lines there) — the logic itself lives in upstreamError.ts, next to the extractErrorMessage it reuses, so nothing else moved into this file. The second line is the repo's own lint-staged prettier pass splitting a pre-existing two-statements-on-one-line at getProviderCredentials (`invalidateManagedLease(...); log.warn(...)`); it re-applies on any commit that touches this file, so it is not separable from the change. Covered by tests/unit/provider-error-detail-lastError.test.ts.",
|
||||
"_rebaseline_2026_08_23_11186_synced_inventory_routing": "PR #11186 (pacocartones) own growth: src/sse/services/auth.ts 3260->3337 (+77, loadAdvertisedModelsForSelfHostedConnections + the modelNotAdvertised candidate-filter predicate — pins chat routing to the connection whose synced inventory actually advertises the model, fixing spurious model-not-found on multi-host self-hosted setups; at the existing credential-selection chokepoint, not extractable without splitting the selection flow). Covered by tests/unit/chat-routing-synced-inventory-11089.test.ts. Owner pre-authorized baseline bumps 2026-08-22.",
|
||||
"tests/unit/account-fallback-service.test.ts": 2044,
|
||||
"tests/unit/provider-validation-specialty.test.ts": 3880,
|
||||
@@ -472,7 +475,10 @@
|
||||
"_rebaseline_2026_08_21_10907_sticky_pin_clear": "#10907 own growth: open-sse/executors/commandCode.ts 1023->1038 (+15, effort-suffix sanitization threading for the sticky-pin-clear fix). Cohesive change at the existing executor chokepoint. Covered by tests/unit/command-code-executor.test.ts.",
|
||||
"_rebaseline_2026_08_21_10986_reasoning_only_content": "#10986 own growth: open-sse/executors/commandCode.ts 1038->1059 (+21, reasoning-only content fallback — when upstream emits only reasoning-delta events and never a text-delta, surface the reasoning text as message.content in createJsonResponse and emit a synthetic content delta in createStreamResponse). Cohesive bug fix at the existing executor chokepoint (mirrors precedent style of #10907/#10859). Covered by tests/unit/command-code-executor.test.ts (2 new cases: non-stream + streaming).",
|
||||
"_rebaseline_2026_08_21_11069_m365_har_import": "#11069 own growth: AddApiKeyModal.tsx 1073->1080 (+7 = Import .har file button for the copilot-m365-web credential modal — M365 is the only provider whose credential (access_token+chathubPath) must be extracted from a DevTools HAR WebSocket URL, added as a new modal affordance). Cohesive UI at the existing modal chokepoint; not extractable. Covered by tests/unit/m365-har-import*.test.ts.",
|
||||
"_rebaseline_2026_08_23_tip_drift_post_batch0823": "Tip drift after the 2026-08-23 merge wave: chatBodyAdmission.ts 1009->1118 (+109, gate count incl. +1) and auth.ts 3337->3344 (+7), both grown by merges already on origin/release/v3.8.50 (verified identical on the pristine tip) — not by the codex-appserver-hardening PR that carries this bump. Owner pre-authorized baseline bumps 2026-08-22."
|
||||
"_rebaseline_2026_08_23_tip_drift_post_batch0823": "Tip drift after the 2026-08-23 merge wave: chatBodyAdmission.ts 1009->1118 (+109, gate count incl. +1) and auth.ts 3337->3344 (+7), both grown by merges already on origin/release/v3.8.50 (verified identical on the pristine tip) — not by the codex-appserver-hardening PR that carries this bump. Owner pre-authorized baseline bumps 2026-08-22.",
|
||||
"_rebaseline_2026_08_24_11355_cooldown_recovery_guards": "PR #11355 own growth: test/route.ts 1215->1237, +22 (startup crash-recovery guard: clearStaleCrashCooldowns() now parses the persisted rate_limited_until deadline and skips clearing rows still genuinely in the future, instead of clearing every non-terminal cooldown unconditionally). Cohesive fix at the existing test-route dispatch chokepoint alongside the #11141 probe builder. Covered by tests/unit/startup-stale-cooldown-recovery.test.ts + tests/unit/repro-zai-cooldown-cleared-by-connection-test.test.ts.",
|
||||
"src/lib/guardrails/videoBridgeRuntime.ts": 1009,
|
||||
"_rebaseline_2026_08_24_video_bridge_fu02_fu07_sampler": "PRs #11344 (FU-02 one-frame scene-aware determinism) + #11381 (FU-07 opt-in segment_aware structural sampling) own growth: videoBridgeRuntime.ts <1000->1009, +9 (sum of both boarded together in the same merge-batch). #11344 adds the deterministic one-frame midpoint fallback + policyEffective=uniform report at the existing scene_aware seam; #11381 adds the bounded local-only FFmpeg structural pre-analysis pass (scene/freeze/blur/exposure/SI-TI) and its budget-reallocation logic. Covered by tests/unit/guardrails/videoBridgeSampler.test.ts, tests/unit/guardrails/videoBridgeFu07StructuralSampling.test.ts, tests/integration/video-bridge-sampler-ffmpeg.test.ts. Owner pre-authorized rebaseline for legitimate PR growth (2026-08-19 directive)."
|
||||
},
|
||||
"_rebaseline_base_2026_08_10_proxyfetch": "Base-red fix (green-prs sweep, issue #9985): open-sse/utils/proxyFetch.ts 1207 > cap 1000 — new proxied-TLS fetch helper introduced by the Fal reference-image work. Owner-authorized quick rebaseline to green; structural slim tracked for v3.9.0.",
|
||||
"_rebaseline_2026_07_27_v3849_train2": "Merge-train 2 (7 PRs) — owner-approved 2026-07-27. Single entry: chatCore.ts 4955->5006 (#8595, Responses multi-turn image compaction before the context hard-reject). Genuine irreducible growth at the existing compaction chokepoint in handleChatCore — the PR adds a last-resort retry against the concrete budget plus the estimateFinalInputTokens helper, both wired at the pre-existing call site rather than a new branch. Covered by tests/unit/8560-responses-image-compaction.test.ts (4 tests).",
|
||||
|
||||
4
config/release/changelog-reconciliations.json
Normal file
@@ -0,0 +1,4 @@
|
||||
{
|
||||
"schemaVersion": 1,
|
||||
"reconciliations": []
|
||||
}
|
||||
@@ -108,24 +108,48 @@ A successful policy returns `AuthSubject` with `kind ∈ { client_api_key, dashb
|
||||
|
||||
`src/shared/constants/publicApiRoutes.ts` is the explicit allowlist:
|
||||
|
||||
The list is split by **shape**, and the split is load-bearing (GHSA-74g9-q8f6-793h): a prefix is
|
||||
matched with `startsWith()`, so it also matches every adjacent path sharing its leading characters.
|
||||
`/api/usage/om-usage` as a prefix marked `/api/usage/om-usage<anything>` PUBLIC, and Next resolves
|
||||
that to `/api/usage/[connectionId]` — a handler with no auth of its own.
|
||||
|
||||
```ts
|
||||
// Genuine subtrees. Every entry MUST end in "/" (asserted by a unit test).
|
||||
PUBLIC_API_ROUTE_PREFIXES = [
|
||||
"/api/auth/oidc/",
|
||||
"/api/v1/", // treated as CLIENT_API in classify, not as "no-auth public"
|
||||
"/api/oauth/",
|
||||
"/api/codex/connect/",
|
||||
"/api/telegram/",
|
||||
"/api/cursor-cli/",
|
||||
];
|
||||
|
||||
// Single routes, matched EXACTLY (with or without a trailing slash).
|
||||
PUBLIC_API_ROUTES_EXACT = new Set([
|
||||
"/api/auth/login",
|
||||
"/api/auth/logout",
|
||||
"/api/auth/status",
|
||||
"/api/init",
|
||||
"/api/v1/", // treated as CLIENT_API in classify, not as "no-auth public"
|
||||
"/api/cloud/",
|
||||
"/api/sync/bundle",
|
||||
"/api/oauth/",
|
||||
"/api/cli/connect",
|
||||
"/api/usage/om-usage",
|
||||
"/api/skills/collect/chaos",
|
||||
]);
|
||||
|
||||
// Read-only single routes that also take the CORS origin relaxation.
|
||||
PUBLIC_READONLY_CORS_API_ROUTES = [
|
||||
"/api/health/ping",
|
||||
"/api/monitoring/health",
|
||||
"/api/settings/require-login",
|
||||
];
|
||||
|
||||
PUBLIC_READONLY_API_ROUTE_PREFIXES = ["/api/monitoring/health", "/api/settings/require-login"];
|
||||
// Read-only single route WITHOUT the CORS relaxation.
|
||||
PUBLIC_READONLY_API_ROUTES_EXACT = new Set(["/api/health"]);
|
||||
|
||||
PUBLIC_READONLY_METHODS = new Set(["GET", "HEAD", "OPTIONS"]);
|
||||
```
|
||||
|
||||
Read-only prefixes are public **only** for safe methods. Note: `classifyRoute()` excludes `/api/v1/*` and `/api/v1beta/*` from the PUBLIC fall-through — those are always `CLIENT_API` so the Bearer-key policy still applies.
|
||||
Read-only routes are public **only** for safe methods. Note: `classifyRoute()` excludes `/api/v1/*` and `/api/v1beta/*` from the PUBLIC fall-through — those are always `CLIENT_API` so the Bearer-key policy still applies.
|
||||
|
||||
## Adding a New Route
|
||||
|
||||
@@ -168,7 +192,7 @@ export async function POST(request: Request) {
|
||||
|
||||
### Pattern 3 — Adding to the public allowlist
|
||||
|
||||
Add the prefix to `PUBLIC_API_ROUTE_PREFIXES` (or `PUBLIC_READONLY_API_ROUTE_PREFIXES` for GET-only). Update unit tests at `tests/unit/public-api-routes.test.ts` and `tests/unit/authz/classify.test.ts`.
|
||||
Pick the set by shape, not by convenience. One route goes in `PUBLIC_API_ROUTES_EXACT` (or `PUBLIC_READONLY_CORS_API_ROUTES` for GET-only); only a genuine subtree goes in `PUBLIC_API_ROUTE_PREFIXES`, and it **must end in `/`**. Putting a single route in the prefix list also publishes every adjacent path that shares its leading characters — including dynamic-segment siblings added later (GHSA-74g9-q8f6-793h). Update unit tests at `tests/unit/public-api-routes.test.ts`, `tests/unit/authz/public-route-exact-match.test.ts` and `tests/unit/authz/classify.test.ts`.
|
||||
|
||||
## Scopes
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ Mermaid sources (`.mmd`) and exported SVGs for OmniRoute v3.8.0 architecture flo
|
||||
| [auto-combo-12factor.mmd](./auto-combo-12factor.mmd) | [SVG](./exported/auto-combo-12factor.svg) | docs/routing/AUTO-COMBO.md |
|
||||
| [resilience-3layers.mmd](./resilience-3layers.mmd) | [SVG](./exported/resilience-3layers.svg) | docs/architecture/RESILIENCE_GUIDE.md, CLAUDE.md |
|
||||
| [i18n-flow.mmd](./i18n-flow.mmd) | [SVG](./exported/i18n-flow.svg) | docs/guides/I18N.md |
|
||||
| [mcp-tools-107.mmd](./mcp-tools-107.mmd) | [SVG](./exported/mcp-tools-107.svg) | docs/frameworks/MCP-SERVER.md |
|
||||
| [mcp-tools-107.mmd](./mcp-tools-107.mmd) | [SVG](./exported/mcp-tools-107.svg) | docs/frameworks/MCP-SERVER.md |
|
||||
| [cloud-agent-flow.mmd](./cloud-agent-flow.mmd) | [SVG](./exported/cloud-agent-flow.svg) | docs/frameworks/CLOUD_AGENT.md |
|
||||
| [authz-pipeline.mmd](./authz-pipeline.mmd) | [SVG](./exported/authz-pipeline.svg) | docs/architecture/AUTHZ_GUIDE.md |
|
||||
| [db-schema-overview.mmd](./db-schema-overview.mmd) | [SVG](./exported/db-schema-overview.svg) | docs/architecture/CODEBASE_DOCUMENTATION.md |
|
||||
@@ -34,11 +34,11 @@ inside GitHub's `<img>` sandbox:
|
||||
| [combo-always-on.svg](./combo-always-on.svg) | style reference | Animated priority-combo fallback (4 layers, 16s loop). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [cli-terminal.svg](./cli-terminal.svg) | README.md (root) | Compact half-height animated terminal (1200×350): 3 real CLI commands cycling with typewriter + scrolling subcommand ticker; first frame = completed providers screen. Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [compression-pipeline.svg](./compression-pipeline.svg) | README.md (root) | Animated 10-engine compression funnel (8s loop). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [free-tier-budget.svg](./free-tier-budget.svg) | README.md (root) | Animated free-tier budget card (~1.53B/mo quantified headline, 19-pool budget bar, per-model grid, signup credits, 10s loop). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [readme-hero.svg](./readme-hero.svg) | README.md (root) | Animated hero card (tagline, live provider/free-access headline, full-width compression bar demo, 6 stat chips). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [free-tier-budget.svg](./free-tier-budget.svg) | README.md (root) | Animated free-tier budget card (~1.51B/mo quantified headline, 20-pool budget bar, per-pool grid, signup credits, 10s loop). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [readme-hero.svg](./readme-hero.svg) | README.md (root) | Animated hero card (tagline, live provider/free-access headline, full-width compression bar demo, 6 stat chips). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [promise-pillars.svg](./promise-pillars.svg) | README.md (root) | Animated "The Promise" 6-pillar card (12s border-highlight sweep). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [why-pain-fix.svg](./why-pain-fix.svg) | README.md (root) | Animated "Why OmniRoute" 10-row pain-vs-fix ledger (15s green row sweep). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [strategies-grid.svg](./strategies-grid.svg) | README.md (root) | Animated grid illustrating 18 of the 19 routing strategies; `cache-optimized` remains documented in the adjacent table. Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [strategies-grid.svg](./strategies-grid.svg) | README.md (root) | Animated grid illustrating 18 of the 19 routing strategies; `cache-optimized` remains documented in the adjacent table. Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [privacy-local.svg](./privacy-local.svg) | README.md (root) | Animated "Private & Local-First" 11-row guarantee ledger with receipt chips (16s green row sweep). Edit the SVG directly — there is no `.mmd` source. |
|
||||
| [resilience-layers.svg](./resilience-layers.svg) | README.md (root) | Animated 3-layer resilience card (breaker states CLOSED→OPEN→HALF-OPEN, key cooldown with ×2 backoff, model lockout — 18s loops). Edit the SVG directly — there is no `.mmd` source. |
|
||||
|
||||
|
||||
@@ -1,24 +1,28 @@
|
||||
%% Auto-Combo 13-factor scoring
|
||||
%% Auto-Combo 15-factor scoring
|
||||
%% Reflects: open-sse/services/autoCombo/scoring.ts (DEFAULT_WEIGHTS, sum = 1.0)
|
||||
%% v3.8.49
|
||||
%% v3.8.50
|
||||
%% svg-title: OmniRoute Auto-Combo 15-factor scoring
|
||||
%% svg-description: Flow from an incoming request through eligible candidates, the 15 weighted scoring factors, descending score sort, top-N selection, and sequential dispatch.
|
||||
flowchart TB
|
||||
Request["Incoming request"] --> Candidates["Eligible candidates<br/>(provider × model × account)"]
|
||||
Candidates --> Score["Compute composite score<br/>per candidate"]
|
||||
|
||||
subgraph Factors["13-factor scoring weights (sum = 1.0)"]
|
||||
f1["health (0.20)"]
|
||||
f2["quota (0.15)"]
|
||||
f3["costInv (0.15)"]
|
||||
f4["latencyInv (0.12)"]
|
||||
f5["taskFit (0.08)"]
|
||||
f6["stability (0.05)"]
|
||||
f7["tierPriority (0.05)"]
|
||||
f8["tierAffinity (0.05)"]
|
||||
f9["specificityMatch (0.05)"]
|
||||
f10["contextAffinity (0.05)"]
|
||||
f11["connectionDensity (0.05)"]
|
||||
f12["cacheAffinity (0.00)"]
|
||||
f13["resetWindowAffinity (0.00)"]
|
||||
subgraph Factors["15-factor scoring weights (sum = 1.0)"]
|
||||
f1["quota (0.1429)"]
|
||||
f2["health (0.1605)"]
|
||||
f3["costInv (0.1429)"]
|
||||
f4["latencyInv (0.1143)"]
|
||||
f5["taskFit (0.0762)"]
|
||||
f6["stability (0.0476)"]
|
||||
f7["tierPriority (0.0476)"]
|
||||
f8["tierAffinity (0.0476)"]
|
||||
f9["specificityMatch (0.0476)"]
|
||||
f10["contextAffinity (0.0476)"]
|
||||
f11["cacheAffinity (0.0000)"]
|
||||
f12["sessionAvailability (0.0476)"]
|
||||
f13["resetWindowAffinity (0.0000)"]
|
||||
f14["connectionDensity (0.0476)"]
|
||||
f15["quality (0.0300)"]
|
||||
end
|
||||
|
||||
Score --> Factors
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
<svg viewBox="0 0 1200 350" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Animated terminal demoing the OmniRoute CLI: omniroute providers list (350 providers registered, anthropic, codex, glm, kimi shown active), omniroute combo list (always-on priority, cost-saver, fusion-panel, context-relay) and omniroute health (healthy, 18412 requests in 24h, p95 412ms, circuit breakers 24 closed, 1 half-open, 0 open), cycling over the 80+ command surface: providers, oauth, keys, combo, nodes, models, cache, compression, cost, usage, quota, health, resilience, telemetry, logs, audit, mcp, a2a, cloud, memory, skills, eval, doctor, repl, tunnel, backup, sync, webhooks, policy, pricing, translator, simulate and more.">
|
||||
<svg viewBox="0 0 1200 350" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="Animated terminal demoing the OmniRoute CLI: omniroute providers list (350 providers registered, anthropic, codex, glm, kimi shown active), omniroute combo list (always-on priority, cost-saver, fusion-panel, context-relay) and omniroute health (healthy, 18412 requests in 24h, p95 412ms, circuit breakers 24 closed, 1 half-open, 0 open), cycling over 85 top-level commands: providers, oauth, keys, combo, nodes, models, cache, compression, cost, usage, quota, health, resilience, telemetry, logs, audit, mcp, a2a, cloud, memory, skills, eval, doctor, repl, tunnel, backup, sync, webhooks, policy, pricing, translator, simulate and more.">
|
||||
<desc>Compact animated terminal cycling three real OmniRoute CLI commands with a typewriter effect and a scrolling subcommand ticker; the first frame shows the completed providers-list screen.</desc>
|
||||
<defs><clipPath id="tickerClip"><rect x="12" y="304" width="1176" height="40"/></clipPath><clipPath id="tw0"><rect x="64" y="46" height="26" width="0"><animate attributeName="width" calcMode="discrete" values="0;31;61;92;122;153;184;214;245;245" keyTimes="0;0.012;0.018;0.024;0.030;0.036;0.042;0.048;0.054;1" dur="18s" repeatCount="indefinite"/></rect></clipPath><clipPath id="tw1"><rect x="64" y="46" height="26" width="0"><animate attributeName="width" calcMode="discrete" values="0;26;51;76;102;128;153;178;204;204" keyTimes="0;0.348;0.351;0.357;0.363;0.369;0.375;0.381;0.387;1" dur="18s" repeatCount="indefinite"/></rect></clipPath><clipPath id="tw2"><rect x="64" y="46" height="26" width="0"><animate attributeName="width" calcMode="discrete" values="0;20;41;61;82;102;122;143;163;163" keyTimes="0;0.678;0.684;0.690;0.696;0.702;0.708;0.714;0.720;1" dur="18s" repeatCount="indefinite"/></rect></clipPath></defs>
|
||||
<rect width="1200" height="350" fill="#0d1117"/>
|
||||
<rect x="0" y="0" width="1200" height="34" fill="#161b22"/>
|
||||
<path d="M 0 34 L 1200 34" stroke="#ffffff" stroke-opacity="0.08" stroke-width="1"/>
|
||||
<circle cx="24" cy="17" r="6" fill="#ff5f56"/><circle cx="46" cy="17" r="6" fill="#ffbd2e"/><circle cx="68" cy="17" r="6" fill="#27c93f"/>
|
||||
<text x="600" y="22" text-anchor="middle" font-family="Consolas, 'Courier New', monospace" font-size="13" fill="#71717a">omniroute — 80+ commands</text>
|
||||
<g font-family="Consolas, 'Courier New', monospace" font-size="17"><animate attributeName="opacity" values="1;0;0" keyTimes="0;0.006;1" dur="18s" repeatCount="indefinite"/><text x="64" y="66" fill="#F7F6FC">omniroute providers list</text><text x="40" y="100" font-weight="700" fill="#38bdf8">OmniRoute Providers</text><text x="40" y="128" fill="#a1a1aa">1f3a9c2e  anthropic   Claude Max 20x    <tspan fill='#22c55e'>active</tspan></text><text x="40" y="154" fill="#a1a1aa">8c2d5b1a  codex       Codex Pro (team)  <tspan fill='#22c55e'>active</tspan></text><text x="40" y="180" fill="#a1a1aa">f4e0a97b  glm         GLM Coding Plan   <tspan fill='#22c55e'>active</tspan></text><text x="40" y="206" fill="#a1a1aa">03bd6e5f  kimi        Kimi K2 free      <tspan fill='#22c55e'>active</tspan></text><text x="40" y="232" fill="#71717a">… 334 more providers</text></g><g opacity="1" font-family="Consolas, 'Courier New', monospace" font-size="17">
|
||||
<text x="600" y="22" text-anchor="middle" font-family="Consolas, 'Courier New', monospace" font-size="13" fill="#71717a">omniroute — 85 top-level commands</text>
|
||||
<g font-family="Consolas, 'Courier New', monospace" font-size="17"><animate attributeName="opacity" values="1;0;0" keyTimes="0;0.006;1" dur="18s" repeatCount="indefinite"/><text x="64" y="66" fill="#F7F6FC">omniroute providers list</text><text x="40" y="100" font-weight="700" fill="#38bdf8">OmniRoute Providers</text><text x="40" y="128" fill="#a1a1aa">1f3a9c2e  anthropic   Claude Max 20x    <tspan fill='#22c55e'>active</tspan></text><text x="40" y="154" fill="#a1a1aa">8c2d5b1a  codex       Codex Pro (team)  <tspan fill='#22c55e'>active</tspan></text><text x="40" y="180" fill="#a1a1aa">f4e0a97b  glm         GLM Coding Plan   <tspan fill='#22c55e'>active</tspan></text><text x="40" y="206" fill="#a1a1aa">03bd6e5f  kimi        Kimi K2 free      <tspan fill='#22c55e'>active</tspan></text><text x="40" y="232" fill="#71717a">… 346 more providers</text></g><g opacity="1" font-family="Consolas, 'Courier New', monospace" font-size="17">
|
||||
<animate attributeName="opacity" values="1;1;0;0" keyTimes="0;0.315;0.33;1" dur="18s" repeatCount="indefinite"/>
|
||||
<text x="40" y="66" fill="#22c55e">$</text>
|
||||
<g clip-path="url(#tw0)"><text x="64" y="66" fill="#F7F6FC">omniroute providers list</text></g>
|
||||
@@ -14,7 +14,7 @@
|
||||
<animate attributeName="x" calcMode="discrete" values="64;95;125;156;186;217;248;278;309;309" keyTimes="0.000;0.012;0.018;0.024;0.030;0.036;0.042;0.048;0.054;1" dur="18s" repeatCount="indefinite"/>
|
||||
<animate attributeName="opacity" values="0;0;1;0.2;1;0.2;1;0;0" keyTimes="0;0.011;0.012;0.022;0.032;0.042;0.052;0.074;1" dur="18s" repeatCount="indefinite"/>
|
||||
</rect>
|
||||
<text x="40" y="100" font-weight="700" fill="#38bdf8" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.045;0.047" dur="18s" repeatCount="indefinite"/>OmniRoute Providers</text><text x="40" y="128" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.053;0.055" dur="18s" repeatCount="indefinite"/>1f3a9c2e  anthropic   Claude Max 20x    <tspan fill='#22c55e'>active</tspan></text><text x="40" y="154" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.061;0.063" dur="18s" repeatCount="indefinite"/>8c2d5b1a  codex       Codex Pro (team)  <tspan fill='#22c55e'>active</tspan></text><text x="40" y="180" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.069;0.07100000000000001" dur="18s" repeatCount="indefinite"/>f4e0a97b  glm         GLM Coding Plan   <tspan fill='#22c55e'>active</tspan></text><text x="40" y="206" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.077;0.079" dur="18s" repeatCount="indefinite"/>03bd6e5f  kimi        Kimi K2 free      <tspan fill='#22c55e'>active</tspan></text><text x="40" y="232" fill="#71717a" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.085;0.08700000000000001" dur="18s" repeatCount="indefinite"/>… 334 more providers</text>
|
||||
<text x="40" y="100" font-weight="700" fill="#38bdf8" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.045;0.047" dur="18s" repeatCount="indefinite"/>OmniRoute Providers</text><text x="40" y="128" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.053;0.055" dur="18s" repeatCount="indefinite"/>1f3a9c2e  anthropic   Claude Max 20x    <tspan fill='#22c55e'>active</tspan></text><text x="40" y="154" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.061;0.063" dur="18s" repeatCount="indefinite"/>8c2d5b1a  codex       Codex Pro (team)  <tspan fill='#22c55e'>active</tspan></text><text x="40" y="180" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.069;0.07100000000000001" dur="18s" repeatCount="indefinite"/>f4e0a97b  glm         GLM Coding Plan   <tspan fill='#22c55e'>active</tspan></text><text x="40" y="206" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.077;0.079" dur="18s" repeatCount="indefinite"/>03bd6e5f  kimi        Kimi K2 free      <tspan fill='#22c55e'>active</tspan></text><text x="40" y="232" fill="#71717a" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.085;0.08700000000000001" dur="18s" repeatCount="indefinite"/>… 346 more providers</text>
|
||||
</g><g opacity="0" font-family="Consolas, 'Courier New', monospace" font-size="17">
|
||||
<animate attributeName="opacity" values="0;0;1;1;0;0" keyTimes="0;0.333;0.34800000000000003;0.648;0.663;1" dur="18s" repeatCount="indefinite"/>
|
||||
<text x="40" y="66" fill="#22c55e">$</text>
|
||||
@@ -32,11 +32,11 @@
|
||||
<animate attributeName="x" calcMode="discrete" values="64;84;105;125;146;166;186;207;227;227" keyTimes="0;0.678;0.684;0.690;0.696;0.702;0.708;0.714;0.720;1" dur="18s" repeatCount="indefinite"/>
|
||||
<animate attributeName="opacity" values="0;0;1;0.2;1;0.2;1;0;0" keyTimes="0;0.677;0.678;0.688;0.698;0.708;0.718;0.74;1" dur="18s" repeatCount="indefinite"/>
|
||||
</rect>
|
||||
<text x="40" y="100" font-weight="700" fill="#38bdf8" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.711;0.713" dur="18s" repeatCount="indefinite"/>OmniRoute Health</text><text x="40" y="128" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.719;0.721" dur="18s" repeatCount="indefinite"/>  Status: <tspan fill='#22c55e'>healthy</tspan>   Uptime: 4d 12h 33m</text><text x="40" y="154" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.727;0.729" dur="18s" repeatCount="indefinite"/>  Requests (24h): 18,412   p95: 412ms</text><text x="40" y="180" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.735;0.737" dur="18s" repeatCount="indefinite"/>  Breakers: <tspan fill='#22c55e'>● 24 closed</tspan>  <tspan fill='#f59e0b'>◒ 1 half-open</tspan>  <tspan fill='#ef4444'>○ 0 open</tspan></text><text x="40" y="206" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.743;0.745" dur="18s" repeatCount="indefinite"/>  Providers: 338 registered   90+ free tiers</text><text x="40" y="232" fill="#71717a" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.751;0.753" dur="18s" repeatCount="indefinite"/>… live: /dashboard · omniroute status</text>
|
||||
<text x="40" y="100" font-weight="700" fill="#38bdf8" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.711;0.713" dur="18s" repeatCount="indefinite"/>OmniRoute Health</text><text x="40" y="128" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.719;0.721" dur="18s" repeatCount="indefinite"/>  Status: <tspan fill='#22c55e'>healthy</tspan>   Uptime: 4d 12h 33m</text><text x="40" y="154" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.727;0.729" dur="18s" repeatCount="indefinite"/>  Requests (24h): 18,412   p95: 412ms</text><text x="40" y="180" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.735;0.737" dur="18s" repeatCount="indefinite"/>  Breakers: <tspan fill='#22c55e'>● 24 closed</tspan>  <tspan fill='#f59e0b'>◒ 1 half-open</tspan>  <tspan fill='#ef4444'>○ 0 open</tspan></text><text x="40" y="206" fill="#a1a1aa" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.743;0.745" dur="18s" repeatCount="indefinite"/>  Providers: 350 registered   90+ free tiers</text><text x="40" y="232" fill="#71717a" opacity="0"><animate attributeName="opacity" calcMode="discrete" values="0;0;1" keyTimes="0;0.751;0.753" dur="18s" repeatCount="indefinite"/>… live: /dashboard · omniroute status</text>
|
||||
</g>
|
||||
<path d="M 0 300 L 1200 300" stroke="#ffffff" stroke-opacity="0.08" stroke-width="1"/>
|
||||
<g clip-path="url(#tickerClip)"><g font-family="Consolas, 'Courier New', monospace" font-size="14" fill="#71717a">
|
||||
<animateTransform attributeName="transform" type="translate" from="0 0" to="-2432 0" dur="55s" repeatCount="indefinite"/>
|
||||
<text x="24" y="330"><tspan fill="#8b5cf6">providers</tspan> · oauth · keys · <tspan fill="#8b5cf6">combo</tspan> · nodes · models · cache · <tspan fill="#8b5cf6">compression</tspan> · cost · usage · quota · <tspan fill="#8b5cf6">health</tspan> · resilience · telemetry · logs · audit · <tspan fill="#8b5cf6">mcp</tspan> · a2a · cloud · <tspan fill="#8b5cf6">memory</tspan> · skills · eval · <tspan fill="#8b5cf6">doctor</tspan> · repl · tunnel · backup · sync · webhooks · policy · pricing · translator · simulate …</text><text x="2456" y="330"><tspan fill="#8b5cf6">providers</tspan> · oauth · keys · <tspan fill="#8b5cf6">combo</tspan> · nodes · models · cache · <tspan fill="#8b5cf6">compression</tspan> · cost · usage · quota · <tspan fill="#8b5cf6">health</tspan> · resilience · telemetry · logs · audit · <tspan fill="#8b5cf6">mcp</tspan> · a2a · cloud · <tspan fill="#8b5cf6">memory</tspan> · skills · eval · <tspan fill="#8b5cf6">doctor</tspan> · repl · tunnel · backup · sync · webhooks · policy · pricing · translator · simulate …</text>
|
||||
</g></g>
|
||||
</svg>
|
||||
</svg>
|
||||
|
||||
|
Before Width: | Height: | Size: 12 KiB After Width: | Height: | Size: 12 KiB |
@@ -23,7 +23,7 @@
|
||||
<g font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif">
|
||||
<g opacity="0"><animate attributeName="opacity" values="0;1" dur="0.4s" begin="0.15s" fill="freeze"/>
|
||||
<text x="44" y="196" font-size="14.5" fill="#c9d1d9">Providers</text>
|
||||
<text x="440" y="196" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="15" font-weight="800" fill="#7ee787">338</text>
|
||||
<text x="440" y="196" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="15" font-weight="800" fill="#7ee787">350</text>
|
||||
<text x="604" y="196" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="13.5" font-weight="600" fill="#8b949e">40+</text>
|
||||
<text x="760" y="196" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="13.5" font-weight="600" fill="#8b949e">400+*</text>
|
||||
<text x="916" y="196" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="13.5" font-weight="600" fill="#8b949e">~5</text>
|
||||
@@ -57,7 +57,7 @@
|
||||
</g>
|
||||
<g opacity="0"><animate attributeName="opacity" values="0;1" dur="0.4s" begin="0.51s" fill="freeze"/>
|
||||
<text x="44" y="364" font-size="14.5" fill="#c9d1d9">Built-in MCP server (own tools)</text>
|
||||
<text x="440" y="364" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="15" font-weight="800" fill="#7ee787">109</text>
|
||||
<text x="440" y="364" text-anchor="middle" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="15" font-weight="800" fill="#7ee787">110</text>
|
||||
<use href="#no" x="604" y="359"/>
|
||||
<use href="#mid" x="760" y="359"/>
|
||||
<use href="#no" x="916" y="359"/>
|
||||
|
||||
|
Before Width: | Height: | Size: 13 KiB After Width: | Height: | Size: 13 KiB |
|
Before Width: | Height: | Size: 25 KiB After Width: | Height: | Size: 26 KiB |
@@ -1,4 +1,5 @@
|
||||
<svg viewBox="0 0 1200 842" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="OmniRoute free-tier budget: about 1.51 billion free tokens per month steady, up to about 2.13 billion in your first month with signup credits, aggregated from the documented free tiers of 40 provider pools and 495 models behind one endpoint, live on /dashboard/free-tiers. Honest pool-deduped math: each shared free pool is counted once — counting every rate limit 24/7 would read about 10B, which we don't publish; 15 providers carry a ToS flag so you decide. Budget bar of the 19 countable free pools with per-model breakdown: Mistral Large 3 1B, GPT-4o mini 150M, Gemini 2.5 Flash 60M, GLM 4.7 30M, Llama 3.3 70B 30M, Grok-3 24M, DeepSeek V4 Pro 20M, GPT-4.1 18M, Llama 4 Scout 15M, GPT-4o 7M, MiniMax-M2.7 6M, Arcee Trinity 5M, and more. First month adds one-time signup credits of about 626M (vertex 300M, agentrouter 200M, predibase 25M, together 25M, glm-cn 20M, doubao 15M, ai21 10M, longcat 10M, deepseek 5M, hyperbolic 5M, nscale 5M). Plus the un-countable: permanently-free no-token-cap providers (SiliconFlow, Z.AI GLM-Flash, Kilo, OpenCode Zen, baidu and more) and a $10 OpenRouter top-up unlocking +24M per month, surfaced separately so they never inflate the headline. Live used/remaining and per-model breakdown on the dashboard.">
|
||||
<svg viewBox="0 0 1200 842" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="OmniRoute free-tier budget: about 1.51 billion free tokens per month steady, up to about 2.13 billion in the first month with signup credits. The catalog contains 455 rows, 448 active and 7 discontinued, grouped into 40 recurring pool keys; 20 pools have a published positive monthly token budget and 20 have a zero, uncapped, or keyless budget. Honest pool-deduped math counts each shared free pool once; 15 providers carry a terms-of-service avoid flag. The 20 quantified pools are Mistral 1 billion, LLM7 150 million, Nara 150 million, Gemini 60 million, Cerebras 30 million, Cloudflare AI 30 million, API Airforce 24 million, Ollama Cloud 20 million, Groq 15 million, Bluesminds 7.2 million, SambaNova 6 million, Arcee 4.8 million, Navy 4.5 million, BazaarLink 3.6 million, OpenRouter 1.2 million, Cohere 800 thousand, HuggingChat 500 thousand, Morph 400 thousand, Hugging Face 200 thousand, and Kiro 25 thousand. One-time signup credits add about 626 million. Uncapped providers and the OpenRouter top-up boost are shown separately so they do not inflate the headline. Live usage remains available at /dashboard/free-tiers.">
|
||||
<desc>Pool-deduplicated chart of the 20 recurring free-token pools with positive published budgets, plus signup credits and uncapped providers shown separately.</desc>
|
||||
<defs>
|
||||
<pattern id="gridPaperF" width="32" height="32" patternUnits="userSpaceOnUse">
|
||||
<path d="M 32 0 L 0 0 0 32" fill="none" stroke="#ffffff" stroke-opacity="0.06" stroke-width="1"/>
|
||||
@@ -63,7 +64,7 @@
|
||||
<text x="60" y="228" font-family="Consolas, 'Courier New', monospace" font-size="104" font-weight="800" fill="url(#gradBrandF)">~1.51B</text>
|
||||
<text x="62" y="266" font-family="Consolas, 'Courier New', monospace" font-size="15" letter-spacing="3" font-weight="700" fill="#a1a1aa">FREE TOKENS / MONTH · <tspan fill="#22c55e">STEADY</tspan></text>
|
||||
<text x="62" y="298" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="16" fill="#F7F6FC">up to <tspan font-weight="800" fill="#22c55e">~2.13B</tspan> in your first month — signup credits</text>
|
||||
<text x="62" y="326" font-family="Consolas, 'Courier New', monospace" font-size="12" fill="#71717a">documented free tiers · <tspan fill="#8b5cf6">40 provider pools</tspan> · <tspan fill="#8b5cf6">495 models</tspan> · one endpoint</text>
|
||||
<text x="62" y="326" font-family="Consolas, 'Courier New', monospace" font-size="12" fill="#71717a">documented free tiers · <tspan fill="#8b5cf6">40 recurring pools</tspan> · <tspan fill="#8b5cf6">455 catalog entries</tspan> · one endpoint</text>
|
||||
|
||||
<!-- ═══ Panel · The honest math ═══ -->
|
||||
<rect x="680" y="84" width="460" height="216" rx="14" fill="#161b22" stroke="#ffffff" stroke-opacity="0.08" stroke-width="1"/>
|
||||
@@ -79,59 +80,61 @@
|
||||
<text x="836" y="244" font-family="Consolas, 'Courier New', monospace" font-size="11.5" fill="#22c55e">counted once ✓</text>
|
||||
<text x="704" y="280" font-family="Consolas, 'Courier New', monospace" font-size="12" fill="#f59e0b"><tspan font-weight="800">15 providers</tspan> ToS-flagged <tspan fill="#71717a">— we flag it · you decide</tspan></text>
|
||||
|
||||
<!-- ═══ Budget bar · 19 countable pools ═══ -->
|
||||
<text x="60" y="356" font-family="Consolas, 'Courier New', monospace" font-size="10.5" letter-spacing="2.5" font-weight="700" fill="#a78bfa">WHERE IT COMES FROM · <tspan fill="#F7F6FC">19 COUNTABLE FREE POOLS</tspan></text>
|
||||
<!-- ═══ Budget bar · 20 quantified recurring pools ═══ -->
|
||||
<text x="60" y="356" font-family="Consolas, 'Courier New', monospace" font-size="10.5" letter-spacing="2.5" font-weight="700" fill="#a78bfa">WHERE IT COMES FROM · <tspan fill="#F7F6FC">20 QUANTIFIED RECURRING POOLS</tspan></text>
|
||||
<g clip-path="url(#barShapeF)">
|
||||
<rect x="60" y="372" width="1080" height="18" fill="#1c2230"/>
|
||||
<g clip-path="url(#barRevF)">
|
||||
<rect x="60.0" y="372" width="662.3" height="18" fill="#6c5ce7"/>
|
||||
<rect x="723.3" y="372" width="106.7" height="18" fill="#00b894"/>
|
||||
<rect x="831.0" y="372" width="47.9" height="18" fill="#0984e3"/>
|
||||
<rect x="879.9" y="372" width="28.3" height="18" fill="#e17055"/>
|
||||
<rect x="909.2" y="372" width="28.3" height="18" fill="#fdcb6e"/>
|
||||
<rect x="938.5" y="372" width="24.4" height="18" fill="#e84393"/>
|
||||
<rect x="963.9" y="372" width="21.8" height="18" fill="#00cec9"/>
|
||||
<rect x="986.7" y="372" width="20.5" height="18" fill="#d63031"/>
|
||||
<rect x="1008.2" y="372" width="18.5" height="18" fill="#a29bfe"/>
|
||||
<rect x="1027.7" y="372" width="13.3" height="18" fill="#55efc4"/>
|
||||
<rect x="1042.0" y="372" width="12.6" height="18" fill="#74b9ff"/>
|
||||
<rect x="1055.6" y="372" width="12.0" height="18" fill="#ffeaa7"/>
|
||||
<rect x="1068.6" y="372" width="11.3" height="18" fill="#fab1a0"/>
|
||||
<rect x="1080.9" y="372" width="9.4" height="18" fill="#81ecec"/>
|
||||
<rect x="1091.3" y="372" width="9.2" height="18" fill="#6c5ce7"/>
|
||||
<rect x="1101.5" y="372" width="9.0" height="18" fill="#00b894"/>
|
||||
<rect x="1111.5" y="372" width="9.0" height="18" fill="#0984e3"/>
|
||||
<rect x="1121.5" y="372" width="8.8" height="18" fill="#e17055"/>
|
||||
<rect x="1131.3" y="372" width="8.7" height="18" fill="#fdcb6e"/>
|
||||
<rect x="60.0" y="372" width="661.4" height="18" fill="#6c5ce7"/>
|
||||
<rect x="722.4" y="372" width="99.2" height="18" fill="#00b894"/>
|
||||
<rect x="822.6" y="372" width="99.2" height="18" fill="#0984e3"/>
|
||||
<rect x="922.9" y="372" width="39.7" height="18" fill="#e17055"/>
|
||||
<rect x="963.5" y="372" width="19.8" height="18" fill="#fdcb6e"/>
|
||||
<rect x="984.4" y="372" width="19.8" height="18" fill="#e84393"/>
|
||||
<rect x="1005.2" y="372" width="15.9" height="18" fill="#00cec9"/>
|
||||
<rect x="1022.1" y="372" width="13.2" height="18" fill="#d63031"/>
|
||||
<rect x="1036.3" y="372" width="9.9" height="18" fill="#a29bfe"/>
|
||||
<rect x="1047.3" y="372" width="7.5" height="18" fill="#55efc4"/>
|
||||
<rect x="1055.8" y="372" width="7.5" height="18" fill="#74b9ff"/>
|
||||
<rect x="1064.3" y="372" width="7.5" height="18" fill="#ffeaa7"/>
|
||||
<rect x="1072.8" y="372" width="7.5" height="18" fill="#fab1a0"/>
|
||||
<rect x="1081.3" y="372" width="7.5" height="18" fill="#81ecec"/>
|
||||
<rect x="1089.9" y="372" width="7.5" height="18" fill="#6c5ce7"/>
|
||||
<rect x="1098.4" y="372" width="7.5" height="18" fill="#00b894"/>
|
||||
<rect x="1106.9" y="372" width="7.5" height="18" fill="#0984e3"/>
|
||||
<rect x="1115.4" y="372" width="7.5" height="18" fill="#e17055"/>
|
||||
<rect x="1124.0" y="372" width="7.5" height="18" fill="#fdcb6e"/>
|
||||
<rect x="1132.5" y="372" width="7.5" height="18" fill="#e84393"/>
|
||||
</g>
|
||||
</g>
|
||||
<circle r="3.2" fill="#F7F6FC">
|
||||
<animateMotion path="M 60,381 L 1140,381" keyPoints="0;0;1;1" keyTimes="0;0.02;0.24;1" calcMode="linear" dur="10s" repeatCount="indefinite"/>
|
||||
<animate attributeName="opacity" values="0;1;1;0;0" keyTimes="0;0.02;0.23;0.26;1" dur="10s" repeatCount="indefinite"/>
|
||||
</circle>
|
||||
<text x="60" y="416" font-family="Consolas, 'Courier New', monospace" font-size="11.5" fill="#71717a">each segment = one free pool · widths floored so every provider shows · honest numbers below</text>
|
||||
<text x="60" y="416" font-family="Consolas, 'Courier New', monospace" font-size="11.5" fill="#71717a">each segment = one recurring pool · widths floored so every pool shows · audited pool budgets below</text>
|
||||
|
||||
<!-- ═══ Per-model grid (19 pools) ═══ -->
|
||||
<!-- ═══ Per-pool grid (20 quantified recurring pools) ═══ -->
|
||||
<g font-family="Consolas, 'Courier New', monospace" font-size="12.5">
|
||||
<circle cx="66" cy="452" r="5" fill="#6c5ce7"/><text x="78" y="456" fill="#c9d1d9">Mistral Large 3 <tspan fill="#71717a">1.00B</tspan></text>
|
||||
<circle cx="346" cy="452" r="5" fill="#00b894"/><text x="358" y="456" fill="#c9d1d9">GPT-4o mini <tspan fill="#71717a">150M</tspan></text>
|
||||
<circle cx="626" cy="452" r="5" fill="#0984e3"/><text x="638" y="456" fill="#c9d1d9">Gemini 2.5 Flash <tspan fill="#71717a">60M</tspan></text>
|
||||
<circle cx="906" cy="452" r="5" fill="#e17055"/><text x="918" y="456" fill="#c9d1d9">GLM 4.7 <tspan fill="#71717a">30M</tspan></text>
|
||||
<circle cx="66" cy="482" r="5" fill="#fdcb6e"/><text x="78" y="486" fill="#c9d1d9">Llama 3.3 70B <tspan fill="#71717a">30M</tspan></text>
|
||||
<circle cx="346" cy="482" r="5" fill="#e84393"/><text x="358" y="486" fill="#c9d1d9">Grok-3 <tspan fill="#71717a">24M</tspan></text>
|
||||
<circle cx="626" cy="482" r="5" fill="#00cec9"/><text x="638" y="486" fill="#c9d1d9">DeepSeek V4 Pro <tspan fill="#71717a">20M</tspan></text>
|
||||
<circle cx="906" cy="482" r="5" fill="#d63031"/><text x="918" y="486" fill="#c9d1d9">GPT-4.1 <tspan fill="#71717a">18M</tspan></text>
|
||||
<circle cx="66" cy="512" r="5" fill="#a29bfe"/><text x="78" y="516" fill="#c9d1d9">Llama 4 Scout <tspan fill="#71717a">15M</tspan></text>
|
||||
<circle cx="346" cy="512" r="5" fill="#55efc4"/><text x="358" y="516" fill="#c9d1d9">GPT-4o <tspan fill="#71717a">7M</tspan></text>
|
||||
<circle cx="626" cy="512" r="5" fill="#74b9ff"/><text x="638" y="516" fill="#c9d1d9">MiniMax-M2.7 <tspan fill="#71717a">6M</tspan></text>
|
||||
<circle cx="906" cy="512" r="5" fill="#ffeaa7"/><text x="918" y="516" fill="#c9d1d9">Arcee Trinity Large Prev <tspan fill="#71717a">5M</tspan></text>
|
||||
<circle cx="66" cy="542" r="5" fill="#fab1a0"/><text x="78" y="546" fill="#c9d1d9">Auto Free <tspan fill="#71717a">4M</tspan></text>
|
||||
<circle cx="346" cy="542" r="5" fill="#81ecec"/><text x="358" y="546" fill="#c9d1d9">Auto <tspan fill="#71717a">1M</tspan></text>
|
||||
<circle cx="626" cy="542" r="5" fill="#6c5ce7"/><text x="638" y="546" fill="#c9d1d9">Command A Reasoning <tspan fill="#71717a">800K</tspan></text>
|
||||
<circle cx="906" cy="542" r="5" fill="#00b894"/><text x="918" y="546" fill="#c9d1d9">ERNIE 4.5 VL 424B <tspan fill="#71717a">500K</tspan></text>
|
||||
<circle cx="66" cy="572" r="5" fill="#0984e3"/><text x="78" y="576" fill="#c9d1d9">morph-v3-large <tspan fill="#71717a">400K</tspan></text>
|
||||
<circle cx="346" cy="572" r="5" fill="#e17055"/><text x="358" y="576" fill="#c9d1d9">Llama 3.1 8B <tspan fill="#71717a">200K</tspan></text>
|
||||
<circle cx="626" cy="572" r="5" fill="#fdcb6e"/><text x="638" y="576" fill="#c9d1d9">Claude Sonnet 4.5 <tspan fill="#71717a">25K</tspan></text>
|
||||
<circle cx="66" cy="452" r="5" fill="#6c5ce7"/><text x="78" y="456" fill="#c9d1d9">Mistral <tspan fill="#71717a">1.00B</tspan></text>
|
||||
<circle cx="346" cy="452" r="5" fill="#00b894"/><text x="358" y="456" fill="#c9d1d9">LLM7 <tspan fill="#71717a">150M</tspan></text>
|
||||
<circle cx="626" cy="452" r="5" fill="#0984e3"/><text x="638" y="456" fill="#c9d1d9">Nara <tspan fill="#71717a">150M</tspan></text>
|
||||
<circle cx="906" cy="452" r="5" fill="#e17055"/><text x="918" y="456" fill="#c9d1d9">Gemini <tspan fill="#71717a">60M</tspan></text>
|
||||
<circle cx="66" cy="482" r="5" fill="#fdcb6e"/><text x="78" y="486" fill="#c9d1d9">Cerebras <tspan fill="#71717a">30M</tspan></text>
|
||||
<circle cx="346" cy="482" r="5" fill="#e84393"/><text x="358" y="486" fill="#c9d1d9">Cloudflare AI <tspan fill="#71717a">30M</tspan></text>
|
||||
<circle cx="626" cy="482" r="5" fill="#00cec9"/><text x="638" y="486" fill="#c9d1d9">API Airforce <tspan fill="#71717a">24M</tspan></text>
|
||||
<circle cx="906" cy="482" r="5" fill="#d63031"/><text x="918" y="486" fill="#c9d1d9">Ollama Cloud <tspan fill="#71717a">20M</tspan></text>
|
||||
<circle cx="66" cy="512" r="5" fill="#a29bfe"/><text x="78" y="516" fill="#c9d1d9">Groq <tspan fill="#71717a">15M</tspan></text>
|
||||
<circle cx="346" cy="512" r="5" fill="#55efc4"/><text x="358" y="516" fill="#c9d1d9">Bluesminds <tspan fill="#71717a">7.2M</tspan></text>
|
||||
<circle cx="626" cy="512" r="5" fill="#74b9ff"/><text x="638" y="516" fill="#c9d1d9">SambaNova <tspan fill="#71717a">6M</tspan></text>
|
||||
<circle cx="906" cy="512" r="5" fill="#ffeaa7"/><text x="918" y="516" fill="#c9d1d9">Arcee <tspan fill="#71717a">4.8M</tspan></text>
|
||||
<circle cx="66" cy="542" r="5" fill="#fab1a0"/><text x="78" y="546" fill="#c9d1d9">Navy <tspan fill="#71717a">4.5M</tspan></text>
|
||||
<circle cx="346" cy="542" r="5" fill="#81ecec"/><text x="358" y="546" fill="#c9d1d9">BazaarLink <tspan fill="#71717a">3.6M</tspan></text>
|
||||
<circle cx="626" cy="542" r="5" fill="#6c5ce7"/><text x="638" y="546" fill="#c9d1d9">OpenRouter <tspan fill="#71717a">1.2M</tspan></text>
|
||||
<circle cx="906" cy="542" r="5" fill="#00b894"/><text x="918" y="546" fill="#c9d1d9">Cohere <tspan fill="#71717a">800K</tspan></text>
|
||||
<circle cx="66" cy="572" r="5" fill="#0984e3"/><text x="78" y="576" fill="#c9d1d9">HuggingChat <tspan fill="#71717a">500K</tspan></text>
|
||||
<circle cx="346" cy="572" r="5" fill="#e17055"/><text x="358" y="576" fill="#c9d1d9">Morph <tspan fill="#71717a">400K</tspan></text>
|
||||
<circle cx="626" cy="572" r="5" fill="#fdcb6e"/><text x="638" y="576" fill="#c9d1d9">Hugging Face <tspan fill="#71717a">200K</tspan></text>
|
||||
<circle cx="906" cy="572" r="5" fill="#e84393"/><text x="918" y="576" fill="#c9d1d9">Kiro <tspan fill="#71717a">25K</tspan></text>
|
||||
</g>
|
||||
|
||||
<!-- ═══ First-month signup credits ═══ -->
|
||||
|
||||
|
Before Width: | Height: | Size: 18 KiB After Width: | Height: | Size: 18 KiB |
@@ -1,4 +1,4 @@
|
||||
<svg viewBox="0 0 1200 540" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="The OmniRoute promise: one endpoint, 350 providers — never stop building, OmniRoute picks the cheapest one that works. Six pillars. Never hit limits: auto-fallback across 350 providers in milliseconds, quota out means the next provider takes over with zero downtime. Save up to 95 percent of tokens: RTK plus Caveman stacked compression cuts 15 to 95 percent of eligible tokens, about 89 percent average on tool-heavy sessions. Zero dollars to start: 90+ providers with a free tier, 56 free forever — Qoder, Pollinations, Cloudflare, SiliconFlow — no card needed. Every tool works: 33 coding agents including Claude Code, Codex, Cursor, Cline, Copilot and Antigravity through one config. One endpoint: OpenAI, Claude, Gemini and Responses API translation — point any tool at /v1 and it just works. Production-grade: circuit breakers, TLS stealth, MCP with 110 tools, A2A, memory, guardrails, evals — 25,000+ tests.">
|
||||
<svg viewBox="0 0 1200 540" xmlns="http://www.w3.org/2000/svg" role="img" aria-label="The OmniRoute promise: one endpoint and 350 providers. Six pillars. Resilient fallback: automatic routing continues while another healthy target is available. Save up to 95 percent of eligible tokens: RTK plus Caveman stacked compression averages about 89 percent on tool-heavy sessions. Zero dollars to start: 90+ providers with a free tier and 56 recurring or keyless free-forever providers. Every tool works: 35 CLI and agent integration records, including Claude Code, Codex, Cursor, Cline, Copilot and Antigravity, through one config. One endpoint: OpenAI, Claude, Gemini and Responses API translation at /v1. Production controls: circuit breakers, TLS stealth, MCP with 110 tools, A2A, memory, guardrails, evals, and 39,000+ static test declarations across 5,100+ tracked test files.">
|
||||
<desc>Animated promise card: six pillar tiles fade in in reading order, then a soft colored border highlight sweeps from tile to tile in a continuous cycle.</desc>
|
||||
<defs>
|
||||
<pattern id="gridPaperP" width="32" height="32" patternUnits="userSpaceOnUse">
|
||||
@@ -40,7 +40,7 @@
|
||||
<text x="102" y="170" font-size="18" font-weight="800" fill="#74b9ff">Never hit limits</text>
|
||||
<text x="66" y="204" font-size="13.5" fill="#a1a1aa">Auto-fallback across 350 providers in</text>
|
||||
<text x="66" y="226" font-size="13.5" fill="#a1a1aa">milliseconds. Quota out? The next provider</text>
|
||||
<text x="66" y="248" font-size="13.5" fill="#a1a1aa">takes over — zero downtime.</text>
|
||||
<text x="66" y="248" font-size="13.5" fill="#a1a1aa">takes over while a healthy target remains.</text>
|
||||
</g>
|
||||
|
||||
<!-- cell 2: save tokens (orange) -->
|
||||
@@ -91,7 +91,7 @@
|
||||
<path d="M 10,18 L 10,22"/>
|
||||
</g>
|
||||
<text x="102" y="354" font-size="18" font-weight="800" fill="#a78bfa">Every tool works</text>
|
||||
<text x="66" y="388" font-size="13.5" fill="#a1a1aa">33 coding agents — Claude Code, Codex,</text>
|
||||
<text x="66" y="388" font-size="13.5" fill="#a1a1aa">35 CLI/agent integrations — Claude Code, Codex,</text>
|
||||
<text x="66" y="410" font-size="13.5" fill="#a1a1aa">Cursor, Cline, Copilot, Antigravity —</text>
|
||||
<text x="66" y="432" font-size="13.5" fill="#a1a1aa">through one config.</text>
|
||||
</g>
|
||||
@@ -127,7 +127,7 @@
|
||||
<text x="862" y="354" font-size="18" font-weight="800" fill="#7ee787">Production-grade</text>
|
||||
<text x="826" y="388" font-size="13.5" fill="#a1a1aa">Circuit breakers, TLS stealth, MCP (110</text>
|
||||
<text x="826" y="410" font-size="13.5" fill="#a1a1aa">tools), A2A, memory, guardrails, evals —</text>
|
||||
<text x="826" y="432" font-size="13.5" fill="#a1a1aa">25,000+ tests.</text>
|
||||
<text x="826" y="432" font-size="13.5" fill="#a1a1aa">39,000+ static test declarations.</text>
|
||||
</g>
|
||||
</g>
|
||||
|
||||
|
||||
|
Before Width: | Height: | Size: 10 KiB After Width: | Height: | Size: 10 KiB |
@@ -66,7 +66,7 @@
|
||||
<!-- stat chips -->
|
||||
<g font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" text-anchor="middle">
|
||||
<rect x="48" y="448" width="172" height="52" rx="12" fill="#161b22" stroke="#6c5ce7" stroke-opacity="0.55" stroke-width="1.5"/>
|
||||
<text x="134" y="471" font-size="17" font-weight="800" fill="#a78bfa">338</text>
|
||||
<text x="134" y="471" font-size="17" font-weight="800" fill="#a78bfa">350</text>
|
||||
<text x="134" y="490" font-size="11" fill="#a1a1aa">AI PROVIDERS</text>
|
||||
<rect x="234" y="448" width="172" height="52" rx="12" fill="#161b22" stroke="#22c55e" stroke-opacity="0.55" stroke-width="1.5"/>
|
||||
<text x="320" y="471" font-size="17" font-weight="800" fill="#7ee787">90+</text>
|
||||
|
||||
|
Before Width: | Height: | Size: 7.3 KiB After Width: | Height: | Size: 7.3 KiB |
|
Before Width: | Height: | Size: 22 KiB After Width: | Height: | Size: 22 KiB |
@@ -95,7 +95,7 @@
|
||||
<rect width="173" height="158" rx="10" fill="#161b22" stroke="#ffffff" stroke-opacity="0.07" stroke-width="1"/>
|
||||
<text x="12" y="21" font-family="Consolas, 'Courier New', monospace" font-size="11" fill="#a78bfa">auto</text>
|
||||
<circle cx="20" cy="79" r="4" fill="none" stroke="#c9d1d9" stroke-width="1.6"/><circle cx="20" cy="79" r="1.6" fill="#c9d1d9"/><path d="M 26,79 C 62,79 84,67.5 112,67.5" fill="none" stroke="#8b5cf6" stroke-opacity="0.55" stroke-width="1.6"/><rect x="116" y="38.0" width="26" height="11" rx="2.5" fill="#1c2330" stroke="#ffffff" stroke-opacity="0.10" stroke-width="1"/><text x="147" y="46.5" font-family="Consolas, 'Courier New', monospace" font-size="8.5" fill="#71717a">72</text><rect x="116" y="62.0" width="26" height="11" rx="2.5" fill="#1c2330" stroke="#7ee787" stroke-opacity="0.8" stroke-width="1"/><text x="147" y="70.5" font-family="Consolas, 'Courier New', monospace" font-size="8.5" fill="#71717a">91</text><rect x="116" y="86.0" width="26" height="11" rx="2.5" fill="#1c2330" stroke="#ffffff" stroke-opacity="0.10" stroke-width="1"/><text x="147" y="94.5" font-family="Consolas, 'Courier New', monospace" font-size="8.5" fill="#71717a">64</text><rect x="116" y="110.0" width="26" height="11" rx="2.5" fill="#1c2330" stroke="#ffffff" stroke-opacity="0.10" stroke-width="1"/><text x="147" y="118.5" font-family="Consolas, 'Courier New', monospace" font-size="8.5" fill="#71717a">55</text><circle r="2.8" fill="#a78bfa" opacity="0"><animateMotion path="M 26,79 C 62,79 84,67.5 110,67.5" begin="3.3s" dur="3.6s" repeatCount="indefinite"/><animate attributeName="opacity" values="0;1;1;0;0" keyTimes="0;0.02;0.3;0.33999999999999997;1" begin="3.3s" dur="3.6s" repeatCount="indefinite"/></circle>
|
||||
<text x="12" y="148" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="9.5" fill="#71717a">live 13-factor scoring</text>
|
||||
<text x="12" y="148" font-family="Inter, 'Segoe UI', Arial, Helvetica, system-ui, sans-serif" font-size="9.5" fill="#71717a">live 15-factor scoring</text>
|
||||
</g><g transform="translate(796,456)">
|
||||
<rect width="173" height="158" rx="10" fill="#161b22" stroke="#ffffff" stroke-opacity="0.07" stroke-width="1"/>
|
||||
<text x="12" y="21" font-family="Consolas, 'Courier New', monospace" font-size="11" fill="#a78bfa">fusion</text>
|
||||
|
||||
|
Before Width: | Height: | Size: 44 KiB After Width: | Height: | Size: 44 KiB |
@@ -1,6 +1,6 @@
|
||||
# Free Tiers Guide: Understand and Combine Free AI Access
|
||||
|
||||
> **TL;DR**: OmniRoute registers 329 providers, with **155 catalog entries marked free/no-auth**. The stricter audited budget currently covers **43 recurring pools / 522 model budget entries**. Connect several suitable providers for broader fallback capacity; every quota, approval rule, privacy policy, and paid-overage condition still applies.
|
||||
> **TL;DR**: OmniRoute registers 350 provider IDs, with **154 provider-catalog entries marked `hasFree`**. The stricter audited free-model catalog covers **40 recurring pool keys / 455 entries** (448 active + 7 discontinued). Connect several suitable providers for broader fallback capacity; every quota, approval rule, privacy policy, and paid-overage condition still applies.
|
||||
|
||||
---
|
||||
|
||||
@@ -21,38 +21,38 @@ OmniRoute **aggregates** these free tiers into one endpoint. Instead of signing
|
||||
|
||||
These providers have a recurring, keyless, or uncapped free-access path in the audited catalog. “Uncapped” means no published token cap; rate, concurrency, account, regional, and policy limits can still apply:
|
||||
|
||||
| Provider | Models | Quota | How to Connect |
|
||||
|----------|--------|-------|----------------|
|
||||
| **Kiro AI** | Claude Sonnet 4.5, Haiku 4.5, DeepSeek V3.2, and others | Audited catalog estimates a 25K-token shared monthly pool | OAuth/account flow; ToS flagged `avoid` in the catalog |
|
||||
| **OpenCode Free** | Current `*-free` model set in the provider registry | Keyless; no published token cap | No provider credential; ToS flagged `avoid` |
|
||||
| **Pollinations** | Current keyless model set; some former models are discontinued or key-required | Keyless; no published token cap | No provider credential for the keyless models |
|
||||
| **Logfare** | kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3, and more | Free API key (no rate limits, no card); **every request is logged** for research (opt out at logfare.ai/consent) | Instant key at logfare.ai/register; ToS/privacy at logfare.ai/tos and logfare.ai/privacy |
|
||||
| **Cloudflare AI** | Workers AI catalog | Audited pool estimates ~30M tokens/month from published usage units | Cloudflare account and API credentials |
|
||||
| **Gemini** | Gemini Flash family | Audited pool estimates ~60M tokens/month | Google AI Studio API key; rate limits apply |
|
||||
| **Groq** | Llama, GPT-OSS, and Qwen models | Audited pool estimates ~15M tokens/month | Groq API key; rate limits apply |
|
||||
| **Cerebras** | GLM 4.7 and GPT-OSS 120B | Audited pool estimates ~30M tokens/month | Cerebras API key; rate limits apply |
|
||||
| Provider | Models | Quota | How to Connect |
|
||||
| ----------------- | ------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------- |
|
||||
| **Kiro AI** | Claude Sonnet 4.5, Haiku 4.5, DeepSeek V3.2, and others | Audited catalog estimates a 25K-token shared monthly pool | OAuth/account flow; ToS flagged `avoid` in the catalog |
|
||||
| **OpenCode Free** | Current `*-free` model set in the provider registry | Keyless; no published token cap | No provider credential; ToS flagged `avoid` |
|
||||
| **Pollinations** | Current keyless model set; some former models are discontinued or key-required | Keyless; no published token cap | No provider credential for the keyless models |
|
||||
| **Logfare** | kimi-k3, deepseek-v4-pro, glm-5.2, gpt-5.6-luna, minimax-m3, and more | Free API key (no rate limits, no card); **every request is logged** for research (opt out at logfare.ai/consent) | Instant key at logfare.ai/register; ToS/privacy at logfare.ai/tos and logfare.ai/privacy |
|
||||
| **Cloudflare AI** | Workers AI catalog | Audited pool estimates ~30M tokens/month from published usage units | Cloudflare account and API credentials |
|
||||
| **Gemini** | Gemini Flash family | Audited pool estimates ~60M tokens/month | Google AI Studio API key; rate limits apply |
|
||||
| **Groq** | Llama, GPT-OSS, and Qwen models | Audited pool estimates ~15M tokens/month | Groq API key; rate limits apply |
|
||||
| **Cerebras** | GLM 4.7 and GPT-OSS 120B | Audited pool estimates ~30M tokens/month | Cerebras API key; rate limits apply |
|
||||
|
||||
### Signup Grants and Provider-Specific Credits
|
||||
|
||||
These providers give you **free credits** when you sign up:
|
||||
|
||||
| Provider | Free Credits | Models | How to Get |
|
||||
|----------|-------------|--------|------------|
|
||||
| **DeepSeek** | 5M free tokens | DeepSeek V4 | Sign up at platform.deepseek.com |
|
||||
| **LongCat** | 10M-token one-time grant | LongCat 2.0 | API key + KYC; pay-as-you-go after the grant |
|
||||
| **Together** | $25 signup credit represented as ~25M tokens in the budget model | Provider catalog | Sign up and verify current terms |
|
||||
| Provider | Free Credits | Models | How to Get |
|
||||
| ------------- | ------------------------------------------------------------------ | ------------------------- | --------------------------------------------------------- |
|
||||
| **DeepSeek** | 5M free tokens | DeepSeek V4 | Sign up at platform.deepseek.com |
|
||||
| **LongCat** | 10M-token one-time grant | LongCat 2.0 | API key + KYC; pay-as-you-go after the grant |
|
||||
| **Together** | $25 signup credit represented as ~25M tokens in the budget model | Provider catalog | Sign up and verify current terms |
|
||||
| **Vertex AI** | $300 signup credit represented as ~300M tokens in the budget model | Gemini and partner models | Google Cloud account; billing and eligibility rules apply |
|
||||
|
||||
### Other Limited Access
|
||||
|
||||
These providers have **free tiers** with specific limits:
|
||||
|
||||
| Provider | Free Limit | Models | Best For |
|
||||
|----------|-----------|--------|----------|
|
||||
| **GitHub Models** | Audited shared pool estimates ~18M tokens/month | Broad model evaluation |
|
||||
| **Hugging Face** | Small recurring monthly pool | Experiments and model variety |
|
||||
| **OpenRouter free models** | Shared request-limited pool; optional one-time top-up increases the recurring allowance | Broad fallback catalog |
|
||||
| **AI Horde** | Keyless community capacity; availability varies | Opportunistic distributed inference |
|
||||
| Provider | Free Limit | Models | Best For |
|
||||
| -------------------------- | --------------------------------------------------------------------------------------- | ----------------------------------- | -------- |
|
||||
| **GitHub Models** | Audited shared pool estimates ~18M tokens/month | Broad model evaluation |
|
||||
| **Hugging Face** | Small recurring monthly pool | Experiments and model variety |
|
||||
| **OpenRouter free models** | Shared request-limited pool; optional one-time top-up increases the recurring allowance | Broad fallback catalog |
|
||||
| **AI Horde** | Keyless community capacity; availability varies | Opportunistic distributed inference |
|
||||
|
||||
---
|
||||
|
||||
@@ -70,6 +70,7 @@ Connect several providers to reduce dependence on any single quota:
|
||||
4. **LongCat** — one-time signup grant (requires KYC)
|
||||
|
||||
Then use `model: "auto"` and OmniRoute will:
|
||||
|
||||
- Try the highest-ranked eligible connection first
|
||||
- If its quota or health check fails → try the next configured provider
|
||||
- If the keyless provider is unavailable → continue through the remaining targets
|
||||
@@ -135,6 +136,7 @@ If one free provider is busy or down, OmniRoute automatically tries the next one
|
||||
### 2. Smart Routing
|
||||
|
||||
OmniRoute picks the **best free provider** for each request based on:
|
||||
|
||||
- Speed — Which provider is fastest right now?
|
||||
- Quality — Which provider is best for this task?
|
||||
- Capacity — Which provider has quota remaining?
|
||||
@@ -157,13 +159,13 @@ provider's quota or access policy.
|
||||
|
||||
The live, pool-deduplicated catalog currently reports:
|
||||
|
||||
| Metric | Current audited value | Interpretation |
|
||||
| --- | ---: | --- |
|
||||
| Recurring quantified grant | **~1.53B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum |
|
||||
| First month with signup grants | **~2.15B tokens** | Recurring total plus one-time and recurring credits |
|
||||
| Quantified inventory | **43 pools / 522 model budget entries** | Budget-model coverage, not the full 329-provider catalog |
|
||||
| Recurring/keyless/uncapped providers represented | **58** | Provider presence in recurring forms of the audited budget catalog |
|
||||
| Free/no-auth discovery entries | **155** | Broader provider metadata; not all have a quantifiable recurring quota |
|
||||
| Metric | Current audited value | Interpretation |
|
||||
| ---------------------------------------------------- | -----------------------------------------------: | ----------------------------------------------------------------------------------------- |
|
||||
| Recurring quantified grant | **~1.51B tokens/month** | Shared pools counted once; excludes uncapped providers from the sum |
|
||||
| First month with signup grants | **~2.13B tokens** | Recurring total plus one-time and recurring credits |
|
||||
| Audited free-model inventory | **40 recurring pool keys / 455 catalog entries** | 448 active + 7 discontinued; distinct from the 350-provider catalog |
|
||||
| Recurring/keyless free-forever providers represented | **56** | Unique providers across recurring daily/monthly/credit/uncapped and keyless catalog types |
|
||||
| Provider catalog entries marked `hasFree` | **154 / 350** | Broader provider metadata; not all have a quantifiable recurring quota |
|
||||
|
||||
These values are computed from `open-sse/config/freeModelCatalog.ts`; see the
|
||||
[Free Tiers Reference](../reference/FREE_TIERS.md) for pool deduplication, ToS flags,
|
||||
|
||||
@@ -219,13 +219,23 @@ docker build --target runner-cli -t omniroute:cli .
|
||||
|
||||
### Build-time resources
|
||||
|
||||
Two build args control what the `builder` stage costs. They are build-time only —
|
||||
Three build args control what the `builder` stage costs. They are build-time only —
|
||||
`OMNIROUTE_MEMORY_MB` (below) is a separate, runtime knob.
|
||||
|
||||
| Build arg | Default | Effect |
|
||||
| --------------------------- | ------- | ---------------------------------------------------------------------- |
|
||||
| `OMNIROUTE_USE_TURBOPACK` | `1` | `0` builds with webpack instead. Lower peak memory, slower. |
|
||||
| `OMNIROUTE_BUILD_MEMORY_MB` | `4096` | V8 heap ceiling (`--max-old-space-size`) for the spawned `next build`. |
|
||||
| Build arg | Default | Effect |
|
||||
| --------------------------- | ------- | ----------------------------------------------------------------------------------- |
|
||||
| `OMNIROUTE_USE_TURBOPACK` | `1` | `0` builds with webpack instead. Lower peak memory, slower. |
|
||||
| `OMNIROUTE_BUILD_MEMORY_MB` | `6144` | V8 heap ceiling (`--max-old-space-size`) for the spawned `next build`. |
|
||||
| `OMNIROUTE_BUILD_WORKERS` | `3` | Feeds `CIRCLE_NODE_TOTAL`; Next derives `workers = N - 1` for page-data collection. |
|
||||
|
||||
`OMNIROUTE_BUILD_WORKERS` is the one to raise on a big builder and the one to
|
||||
suspect when a constrained build dies **after** `✓ Compiled successfully`. Each
|
||||
page-data worker is its own process and inherits `NODE_OPTIONS`, so the heap
|
||||
ceiling is per process, not per build: the default of `3` (→ 2 workers) is sized
|
||||
for the 16 GB / 4 vCPU GitHub-hosted runners the publish pipeline uses. At `8`
|
||||
(→ 7 workers) that runner ran out of memory and buildkit failed the step with
|
||||
`ResourceExhausted: ... cannot allocate memory`. `tests/unit/docker-build-memory-budget.test.ts`
|
||||
does the arithmetic and fails if either knob outgrows the runner.
|
||||
|
||||
Turbopack compiles in native Rust memory that lives **outside** the V8 heap, so
|
||||
`OMNIROUTE_BUILD_MEMORY_MB` does not bound it. On a host with a memory ceiling the
|
||||
@@ -268,12 +278,12 @@ The 1 GiB Docker default is a dashboard/light-chat floor, not a production siz
|
||||
|
||||
Size **cgroup `--memory` above the heap** — native buffers, SQLite, and compression intermediates sit outside V8.
|
||||
|
||||
| Workload | `OMNIROUTE_MEMORY_MB` | Container / cgroup | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| Dashboard, one light chat | `1024` (image default) | ≥2 GiB | |
|
||||
| One coding agent (Claude/Codex/Grok) | `8192` | ≥10 GiB | Typical single-session `/v1/responses` |
|
||||
| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 GiB | Measured V8 abort at ~12 GiB heap |
|
||||
| Three+ concurrent long contexts | do not on one process | serialize / more RAM | Default heavyweight admission is 1 in-flight; raising it without RAM reintroduces the abort |
|
||||
| Workload | `OMNIROUTE_MEMORY_MB` | Container / cgroup | Notes |
|
||||
| ------------------------------------ | ---------------------- | -------------------- | ------------------------------------------------------------------------------------------- |
|
||||
| Dashboard, one light chat | `1024` (image default) | ≥2 GiB | |
|
||||
| One coding agent (Claude/Codex/Grok) | `8192` | ≥10 GiB | Typical single-session `/v1/responses` |
|
||||
| Two concurrent long `/v1/responses` | `10240`–`12288` | ≥12–16 GiB | Measured V8 abort at ~12 GiB heap |
|
||||
| Three+ concurrent long contexts | do not on one process | serialize / more RAM | Default heavyweight admission is 1 in-flight; raising it without RAM reintroduces the abort |
|
||||
|
||||
`omniroute serve` on bare metal calibrates ~35% of RAM (clamped `[512, 4096]`) when `OMNIROUTE_MEMORY_MB` is **unset**. Docker always sets `1024`, so that calibration never runs in the official image.
|
||||
|
||||
@@ -287,19 +297,19 @@ docker run -d --name omniroute --restart unless-stopped --stop-timeout 40 \
|
||||
|
||||
Beyond the defaults documented in [ENVIRONMENT.md](../reference/ENVIRONMENT.md), the following variables matter most when running under Docker:
|
||||
|
||||
| Variable | Purpose | Default |
|
||||
| ----------------------------- | --------------------------------------------------------------------------------------------------- | ------------------------ |
|
||||
| `OMNIROUTE_WS_BRIDGE_SECRET` | Shared secret for the WebSocket bridge. **Required in production** — set to a strong random string. | unset (must be provided) |
|
||||
| `REDIS_URL` | Connection string for the rate limiter / cache backend | `redis://redis:6379` |
|
||||
| `REDIS_PORT` | Host-side port for the bundled Redis container | `6379` |
|
||||
| `REDIS_BIND_HOST` | Host interface the bundled Redis port is published on (loopback unless you add AUTH) | `127.0.0.1` |
|
||||
| `AUTO_UPDATE_HOST_REPO_DIR` | Host path mounted into `cli` profile at `/workspace/omniroute` for self-update workflows | `.` (current directory) |
|
||||
| Variable | Purpose | Default |
|
||||
| ----------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------ |
|
||||
| `OMNIROUTE_WS_BRIDGE_SECRET` | Shared secret for the WebSocket bridge. **Required in production** — set to a strong random string. | unset (must be provided) |
|
||||
| `REDIS_URL` | Connection string for the rate limiter / cache backend | `redis://redis:6379` |
|
||||
| `REDIS_PORT` | Host-side port for the bundled Redis container | `6379` |
|
||||
| `REDIS_BIND_HOST` | Host interface the bundled Redis port is published on (loopback unless you add AUTH) | `127.0.0.1` |
|
||||
| `AUTO_UPDATE_HOST_REPO_DIR` | Host path mounted into `cli` profile at `/workspace/omniroute` for self-update workflows | `.` (current directory) |
|
||||
| `OMNIROUTE_MEMORY_MB` | Runtime Node heap ceiling for the Docker standalone server; overrides the image default above. Coding agents: `8192`+ (see [runtime RAM](#runtime-ram-for-coding-agents)). | `1024` |
|
||||
| `DASHBOARD_PORT` / `API_PORT` | Override exposed ports for dashboard (20128) and API (20129) | `20128` / `20129` |
|
||||
| `OMNIROUTE_BASE_PATH` | URL subpath when the app is published behind a reverse proxy (e.g. `/omniroute`) | _(empty = root)_ |
|
||||
| `NEXT_PUBLIC_BASE_URL` | Public browser origin including the subpath (e.g. `https://host/omniroute`) | unset |
|
||||
| `PROD_DASHBOARD_PORT` | Host-side dashboard port for `docker-compose.prod.yml` | `20130` |
|
||||
| `CLIPROXYAPI_PORT` | Host-side port for the `cliproxyapi` sidecar | `8317` |
|
||||
| `DASHBOARD_PORT` / `API_PORT` | Override exposed ports for dashboard (20128) and API (20129) | `20128` / `20129` |
|
||||
| `OMNIROUTE_BASE_PATH` | URL subpath when the app is published behind a reverse proxy (e.g. `/omniroute`) | _(empty = root)_ |
|
||||
| `NEXT_PUBLIC_BASE_URL` | Public browser origin including the subpath (e.g. `https://host/omniroute`) | unset |
|
||||
| `PROD_DASHBOARD_PORT` | Host-side dashboard port for `docker-compose.prod.yml` | `20130` |
|
||||
| `CLIPROXYAPI_PORT` | Host-side port for the `cliproxyapi` sidecar | `8317` |
|
||||
|
||||
## Reverse Proxy on a Subpath (Traefik / nginx)
|
||||
|
||||
@@ -361,11 +371,11 @@ intervals.
|
||||
|
||||
For orchestrators (Kubernetes, Nomad, etc.):
|
||||
|
||||
| Probe | Prefer | Avoid |
|
||||
| --- | --- | --- |
|
||||
| Liveness | HTTP `GET /livez`, or TCP on the main port (`PORT`, default `20128`) | `/api/monitoring/health` as liveness |
|
||||
| Readiness | HTTP `GET /healthz` | Tight timeouts that treat event-loop busy as dead |
|
||||
| Deep / blackbox | `/api/monitoring/health` | — |
|
||||
| Probe | Prefer | Avoid |
|
||||
| --------------- | -------------------------------------------------------------------- | ------------------------------------------------- |
|
||||
| Liveness | HTTP `GET /livez`, or TCP on the main port (`PORT`, default `20128`) | `/api/monitoring/health` as liveness |
|
||||
| Readiness | HTTP `GET /healthz` | Tight timeouts that treat event-loop busy as dead |
|
||||
| Deep / blackbox | `/api/monitoring/health` | — |
|
||||
|
||||
`/healthz` reports process lifecycle (`ok` / `starting` / `stopping`). `/livez` is
|
||||
process-alive only (200 whenever the handler can run; it does not wait for
|
||||
@@ -431,10 +441,10 @@ Endpoint tunnel panels (Cloudflare, Tailscale, ngrok) can be shown or hidden fro
|
||||
|
||||
## Image Tags
|
||||
|
||||
| Image | Tag | Size | Description |
|
||||
| ------------------------ | -------- | ------ | --------------------- |
|
||||
| Image | Tag | Size | Description |
|
||||
| ------------------------ | -------- | ------ | ---------------------------------------------------- |
|
||||
| `diegosouzapw/omniroute` | `latest` | ~250MB | Highest **published** stable SemVer (not git `main`) |
|
||||
| `diegosouzapw/omniroute` | `3.8.0` | ~250MB | Pin this class of tag for GitOps |
|
||||
| `diegosouzapw/omniroute` | `3.8.0` | ~250MB | Pin this class of tag for GitOps |
|
||||
|
||||
Multi-platform manifest: `linux/amd64` + `linux/arm64` native (Apple Silicon, AWS Graviton, Raspberry Pi). Docker selects the matching architecture automatically; pass `--platform linux/amd64` if you need to force AMD64 emulation on ARM hosts.
|
||||
|
||||
@@ -442,12 +452,12 @@ Multi-platform manifest: `linux/amd64` + `linux/arm64` native (Apple Silicon, AW
|
||||
|
||||
OmniRoute publishes separate Docker channels for stable releases, active release-branch testing, and development builds.
|
||||
|
||||
| Channel | Source | Mutability | Recommended use |
|
||||
| ------------------------------- | ----------------------------------- | --------------------------- | ----------------------------------------------------------------------------------------------- |
|
||||
| `:<version>` / `:<version>-web` | Signed/versioned release | Immutable | Production deployments that pin an exact release |
|
||||
| Channel | Source | Mutability | Recommended use |
|
||||
| ------------------------------- | ----------------------------------- | --------------------------- | --------------------------------------------------------------------------------------------------------------------- |
|
||||
| `:<version>` / `:<version>-web` | Signed/versioned release | Immutable | Production deployments that pin an exact release |
|
||||
| `:latest` / `:latest-web` | Highest **published** stable SemVer | Mutable stable pointer | Follows stable releases **after** a SemVer publish job — does **not** track `main` or unreleased `release/v*` commits |
|
||||
| `:next` / `:next-web` | Current default `release/v*` branch | Mutable pre-release pointer | Testing fixes that have landed on the active release branch but are not yet in a stable release |
|
||||
| `:main` / `:main-web` | `main` branch | Mutable development pointer | Development and integration testing only |
|
||||
| `:next` / `:next-web` | Current default `release/v*` branch | Mutable pre-release pointer | Testing fixes that have landed on the active release branch but are not yet in a stable release |
|
||||
| `:main` / `:main-web` | `main` branch | Mutable development pointer | Development and integration testing only |
|
||||
|
||||
#### Using the pre-release channel
|
||||
|
||||
@@ -491,30 +501,30 @@ A release-branch build can never move `latest`; only an eligible stable semantic
|
||||
|
||||
**`latest` is not a currency guarantee for git.** Merged fixes on `main` or on the active `release/v*` branch are **not** in `:latest` until a stable SemVer image is published and the publish job promotes `:latest` (same digest as that SemVer). If `latest` looks frozen while GitHub already shows the fix, pull `:next` to test the release branch or wait for the SemVer tag.
|
||||
|
||||
| You want | Use |
|
||||
| --- | --- |
|
||||
| GitOps / production that must not drift | Pin `:X.Y.Z` (or the image digest) |
|
||||
| Follow published stables and accept a recreate on each release | `:latest` |
|
||||
| Test unreleased `release/v*` commits | `:next` (not production) |
|
||||
| Test `main` | `:main` (not production) |
|
||||
| You want | Use |
|
||||
| -------------------------------------------------------------- | ---------------------------------- |
|
||||
| GitOps / production that must not drift | Pin `:X.Y.Z` (or the image digest) |
|
||||
| Follow published stables and accept a recreate on each release | `:latest` |
|
||||
| Test unreleased `release/v*` commits | `:next` (not production) |
|
||||
| Test `main` | `:main` (not production) |
|
||||
|
||||
## Availability: default SQLite is single-replica
|
||||
|
||||
Stock Docker / Kubernetes OmniRoute is **one Node process + one SQLite writer**. High availability is **not supported** on that topology.
|
||||
|
||||
| Constraint | Consequence |
|
||||
| --- | --- |
|
||||
| Single writer | Do **not** run multiple replicas against the same SQLite file. That corrupts the DB. |
|
||||
| Constraint | Consequence |
|
||||
| ------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| Single writer | Do **not** run multiple replicas against the same SQLite file. That corrupts the DB. |
|
||||
| Recreate / restart / HEALTHCHECK kill | **Full outage** of in-flight SSE, dashboard sessions, and in-memory state. Every connected client drops. New requests during the empty-endpoint window get a reverse-proxy **`502 Bad Gateway: Unknown error`**, not OmniRoute JSON — clients cannot distinguish this from a provider failure (#11015). |
|
||||
| Same event loop as `/healthz` | A busy catalog or compression tick can delay probes; a short timeout then restarts the **only** replica. |
|
||||
| Same event loop as `/healthz` | A busy catalog or compression tick can delay probes; a short timeout then restarts the **only** replica. |
|
||||
|
||||
**Probe matrix** (see also [Kubernetes probe recommendations](../ops/MONITORING_GUIDE.md#kubernetes-probe-recommendations)):
|
||||
|
||||
| Probe | Target | Do not use |
|
||||
| --- | --- | --- |
|
||||
| Liveness | TCP on `PORT` (default `20128`), or soft HTTP `/healthz` | `/api/monitoring/health` |
|
||||
| Readiness | HTTP `GET /healthz` | Tight timeouts that treat event-loop busy as dead |
|
||||
| Deep / humans | `/api/monitoring/health` | Automated kubelet liveness |
|
||||
| Probe | Target | Do not use |
|
||||
| ------------- | -------------------------------------------------------- | ------------------------------------------------- |
|
||||
| Liveness | TCP on `PORT` (default `20128`), or soft HTTP `/healthz` | `/api/monitoring/health` |
|
||||
| Readiness | HTTP `GET /healthz` | Tight timeouts that treat event-loop busy as dead |
|
||||
| Deep / humans | `/api/monitoring/health` | Automated kubelet liveness |
|
||||
|
||||
**Upgrades:** expect every session to drop. Drain clients if you can; there is no rolling update on default SQLite. Compose `restart: unless-stopped` plus Docker `HEALTHCHECK` will also replace the only process when the container is Unhealthy — same blast radius.
|
||||
|
||||
@@ -555,13 +565,13 @@ One Node process is **one V8 heap**. Two overlapping ~3 MiB / ~750k-token codi
|
||||
|
||||
To go beyond two concurrent **large** jobs **today**:
|
||||
|
||||
| Do | Do not |
|
||||
| --- | --- |
|
||||
| Run **N containers/pods**, each with its **own** `DATA_DIR` / volume | Set `replicas > 1` against one SQLite file |
|
||||
| Keep each instance at 1–2 heavy in-flight and 12–16 Gi cgroup | Give one process 8× RAM and `max=8` |
|
||||
| Optional: `QUOTA_STORE_DRIVER=redis` + `QUOTA_STORE_REDIS_URL` for **shared quota counters** | Treat Redis as shared SQLite — it is not |
|
||||
| Duplicate provider secrets into each instance (or accept partitioned dashboards) | Expect one dashboard / one call-log across instances |
|
||||
| Front with any load balancer; sticky by API key or session is enough | Require a vendor-specific size-aware middleware |
|
||||
| Do | Do not |
|
||||
| -------------------------------------------------------------------------------------------- | ---------------------------------------------------- |
|
||||
| Run **N containers/pods**, each with its **own** `DATA_DIR` / volume | Set `replicas > 1` against one SQLite file |
|
||||
| Keep each instance at 1–2 heavy in-flight and 12–16 Gi cgroup | Give one process 8× RAM and `max=8` |
|
||||
| Optional: `QUOTA_STORE_DRIVER=redis` + `QUOTA_STORE_REDIS_URL` for **shared quota counters** | Treat Redis as shared SQLite — it is not |
|
||||
| Duplicate provider secrets into each instance (or accept partitioned dashboards) | Expect one dashboard / one call-log across instances |
|
||||
| Front with any load balancer; sticky by API key or session is enough | Require a vendor-specific size-aware middleware |
|
||||
|
||||
Hardware: `concurrent_large ≈ N × 2` at ~8–12 Gi heap / ~12–16 Gi cgroup **per instance**. Host RAM must cover `N × cgroup`, not “one 16 Gi pod with N=8.”
|
||||
|
||||
|
||||
@@ -5719,17 +5719,28 @@ paths:
|
||||
x-loopback-only: true
|
||||
tags: [System]
|
||||
summary: Read a bounded Video Bridge drill-down slice
|
||||
description: Internal loopback/token-authenticated lookup into a short-lived per-session frame cache. It never downloads media or starts a subprocess; start/end and frame count only select already materialized frames.
|
||||
description: Internal loopback/token-authenticated lookup into a short-lived cache isolated by an opaque principal, session, and media reference. It never downloads media or starts a subprocess; start/end and frame count only select already materialized, canonicalized JPEG frames whose dimensions were derived from their bytes. This cache substrate is not yet wired to the transparent Video Bridge request path and does not yet expose multi-resolution selection.
|
||||
security: []
|
||||
parameters:
|
||||
- in: header
|
||||
name: x-omniroute-video-bridge-principal
|
||||
required: true
|
||||
description: Canonical visible-ASCII, opaque non-secret principal ID; production tenant derivation is required before enabling a caller
|
||||
schema:
|
||||
type: string
|
||||
minLength: 1
|
||||
maxLength: 256
|
||||
pattern: "^[!-~]{1,256}$"
|
||||
- in: query
|
||||
name: sessionId
|
||||
required: true
|
||||
schema: { type: string, maxLength: 128 }
|
||||
description: Canonical opaque ID without surrounding whitespace
|
||||
schema: { type: string, minLength: 1, maxLength: 128 }
|
||||
- in: query
|
||||
name: videoRef
|
||||
required: true
|
||||
schema: { type: string, maxLength: 4096 }
|
||||
description: Canonical opaque reference without surrounding whitespace
|
||||
schema: { type: string, minLength: 1, maxLength: 4096 }
|
||||
- in: query
|
||||
name: start
|
||||
required: false
|
||||
@@ -5743,25 +5754,58 @@ paths:
|
||||
required: false
|
||||
schema: { type: integer, minimum: 1, maximum: 16 }
|
||||
responses:
|
||||
"200": { description: Bounded cached frame slice }
|
||||
"403": { description: Trusted loopback/token identity required }
|
||||
"200": { description: Bounded cached frame slice with derivation audit metadata }
|
||||
"403": { description: Trusted loopback/token identity and principal required }
|
||||
"404": { description: Drill-down session or media key was not found }
|
||||
post:
|
||||
x-loopback-only: true
|
||||
tags: [System]
|
||||
summary: Store a bounded Video Bridge drill-down result
|
||||
description: Internal lifecycle operation for explicitly authorized callers. The short-lived session cache is isolated by session and media reference and does not alter the primary request cost.
|
||||
description: Internal lifecycle operation for explicitly authorized callers. The short-lived cache is isolated by principal, session, and media reference; enforces independent per-principal and global retained-byte quotas; accepts canonical Base64 only after a warning-sensitive bounded full JPEG decode/re-encode; strips trailing polyglot bytes; retains and charges only the canonical JPEG output; derives resolution from decoded bytes; and does not alter the primary request cost. The JSON wire budget includes Base64 overhead for the 32 MiB decoded-input ceiling.
|
||||
security: []
|
||||
parameters:
|
||||
- in: header
|
||||
name: x-omniroute-video-bridge-principal
|
||||
required: true
|
||||
description: Canonical visible-ASCII, opaque non-secret principal ID; production tenant derivation is required before enabling a caller
|
||||
schema:
|
||||
type: string
|
||||
minLength: 1
|
||||
maxLength: 256
|
||||
pattern: "^[!-~]{1,256}$"
|
||||
requestBody:
|
||||
required: true
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
type: object
|
||||
required: [sessionId, videoRef, durationSeconds, frames]
|
||||
additionalProperties: false
|
||||
required: [sessionId, videoRef, derivation, durationSeconds, frames]
|
||||
properties:
|
||||
sessionId: { type: string, maxLength: 128 }
|
||||
videoRef: { type: string, maxLength: 4096 }
|
||||
sessionId:
|
||||
type: string
|
||||
minLength: 1
|
||||
maxLength: 128
|
||||
description: Canonical opaque ID without surrounding whitespace
|
||||
videoRef:
|
||||
type: string
|
||||
minLength: 1
|
||||
maxLength: 4096
|
||||
description: Canonical opaque reference without surrounding whitespace
|
||||
derivation:
|
||||
type: object
|
||||
additionalProperties: false
|
||||
required: [parentContentHash, policy, version]
|
||||
properties:
|
||||
parentContentHash:
|
||||
type: string
|
||||
pattern: "^sha256:[a-f0-9]{64}$"
|
||||
policy:
|
||||
type: string
|
||||
pattern: "^[A-Za-z0-9][A-Za-z0-9._/-]{0,63}$"
|
||||
version:
|
||||
type: string
|
||||
pattern: "^[A-Za-z0-9][A-Za-z0-9._/-]{0,63}$"
|
||||
durationSeconds: { type: number, exclusiveMinimum: 0, maximum: 600 }
|
||||
frames:
|
||||
type: array
|
||||
@@ -5769,27 +5813,43 @@ paths:
|
||||
maxItems: 16
|
||||
items:
|
||||
type: object
|
||||
additionalProperties: false
|
||||
required: [timestampSeconds, dataUri]
|
||||
properties:
|
||||
timestampSeconds: { type: number, minimum: 0 }
|
||||
dataUri: { type: string, pattern: "^data:image/jpeg;base64," }
|
||||
dataUri:
|
||||
type: string
|
||||
minLength: 27
|
||||
maxLength: 5592431
|
||||
description: Canonical Base64 data URI whose decoded bytes pass a warning-sensitive bounded full JPEG decode/re-encode; trailing bytes are discarded and width and height are derived server-side
|
||||
responses:
|
||||
"201": { description: Drill-down result stored }
|
||||
"403": { description: Trusted loopback/token identity required }
|
||||
"403": { description: Trusted loopback/token identity and principal required }
|
||||
"413": { description: Payload exceeds the bounded session budget }
|
||||
"499": { description: Caller cancelled before the derivation was committed }
|
||||
delete:
|
||||
x-loopback-only: true
|
||||
tags: [System]
|
||||
summary: Delete a Video Bridge drill-down session
|
||||
security: []
|
||||
parameters:
|
||||
- in: header
|
||||
name: x-omniroute-video-bridge-principal
|
||||
required: true
|
||||
description: Canonical visible-ASCII, opaque non-secret principal ID; production tenant derivation is required before enabling a caller
|
||||
schema:
|
||||
type: string
|
||||
minLength: 1
|
||||
maxLength: 256
|
||||
pattern: "^[!-~]{1,256}$"
|
||||
- in: query
|
||||
name: sessionId
|
||||
required: true
|
||||
schema: { type: string, maxLength: 128 }
|
||||
description: Canonical opaque ID without surrounding whitespace
|
||||
schema: { type: string, minLength: 1, maxLength: 128 }
|
||||
responses:
|
||||
"200": { description: Session entries removed }
|
||||
"403": { description: Trusted loopback/token identity required }
|
||||
"403": { description: Trusted loopback/token identity and principal required }
|
||||
|
||||
/api/cache/stats:
|
||||
get:
|
||||
@@ -6931,6 +6991,104 @@ paths:
|
||||
"500":
|
||||
description: Failed to parse OpenAPI spec
|
||||
|
||||
/api/openapi/try:
|
||||
post:
|
||||
tags: [System]
|
||||
summary: Proxy an API Explorer request to an OmniRoute endpoint
|
||||
description: >-
|
||||
Executes an API Explorer request through a server-side, same-origin proxy. The target
|
||||
must start with `/api/`, `/v1/`, `/v1beta/`, `/a2a`, or
|
||||
`/.well-known/agent.json`; protocol-relative and cross-origin targets are rejected.
|
||||
Hop-by-hop, proxy, host, cookie, and forwarding headers supplied in `headers` are
|
||||
stripped, while any dashboard cookie on the original request is forwarded separately.
|
||||
When `requireLogin` is disabled, the management-auth bypass mirrors the runtime setting;
|
||||
otherwise a management Bearer credential or dashboard session is required. Failures
|
||||
caught after authentication, including request JSON parsing, fetch, and response-body
|
||||
parsing failures, are returned in the normal HTTP 200 result envelope so the Explorer
|
||||
can display them; `status: 0` identifies that caught-failure path.
|
||||
security:
|
||||
- BearerAuth: []
|
||||
- ManagementSessionAuth: []
|
||||
requestBody:
|
||||
required: true
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
type: object
|
||||
required: [path]
|
||||
properties:
|
||||
method:
|
||||
type: string
|
||||
enum: [GET, POST, PUT, PATCH, DELETE, HEAD, OPTIONS]
|
||||
default: GET
|
||||
path:
|
||||
type: string
|
||||
minLength: 1
|
||||
pattern: "^/(?:api/|v1/|v1beta/|a2a|\\.well-known/agent\\.json)"
|
||||
description: Same-origin OmniRoute API path, optionally including a query string.
|
||||
headers:
|
||||
type: object
|
||||
default: {}
|
||||
additionalProperties:
|
||||
type: string
|
||||
description: >-
|
||||
Headers to forward after removing connection, content-length, cookie, host,
|
||||
keep-alive, proxy-authenticate, proxy-authorization, te, trailer,
|
||||
transfer-encoding, upgrade, x-forwarded-for, x-forwarded-host, and
|
||||
x-forwarded-proto headers.
|
||||
body:
|
||||
description: >-
|
||||
Optional JSON value. A truthy value is serialized unless it is already a
|
||||
string, and is not forwarded when `method` is `GET`.
|
||||
responses:
|
||||
"200":
|
||||
description: Upstream response or displayable caught-failure envelope
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
type: object
|
||||
additionalProperties: false
|
||||
required: [status, statusText, headers, body, latencyMs, contentType]
|
||||
properties:
|
||||
status:
|
||||
type: integer
|
||||
minimum: 0
|
||||
description: Upstream HTTP status, or 0 when request processing throws.
|
||||
statusText:
|
||||
type: string
|
||||
headers:
|
||||
type: object
|
||||
additionalProperties:
|
||||
type: string
|
||||
body:
|
||||
description: >-
|
||||
Parsed JSON, response text truncated after 10,000 characters, or a sanitized
|
||||
caught-error object.
|
||||
latencyMs:
|
||||
type: integer
|
||||
minimum: 0
|
||||
contentType:
|
||||
type: string
|
||||
"400":
|
||||
description: Invalid request body or non-same-origin path
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
oneOf:
|
||||
- $ref: "#/components/schemas/ValidationErrorResponse"
|
||||
- type: object
|
||||
required: [error]
|
||||
properties:
|
||||
error:
|
||||
type: string
|
||||
example: Path must be same-origin
|
||||
"401":
|
||||
$ref: "#/components/responses/ManagementAuthenticationRequired"
|
||||
"403":
|
||||
$ref: "#/components/responses/ManagementInvalidToken"
|
||||
"503":
|
||||
$ref: "#/components/responses/InternalError"
|
||||
|
||||
# ─── Agent Skills Catalog ────────────────────────────────────────────────────
|
||||
|
||||
/api/agent-skills:
|
||||
|
||||
@@ -531,6 +531,9 @@ detection above).
|
||||
| `OMNIROUTE_CONFIG_HOT_RELOAD_MS` | `5000` | `src/lib/config/hotReload.ts` | Polling interval (ms) for config hot-reload. Lower than `1000` is rejected. |
|
||||
| `OMNIROUTE_DISABLE_REDIS_AUTH_CACHE` | _(enabled)_ | `src/lib/db/apiKeys.ts` | Set `1` to bypass the Redis-backed API-key auth cache (forces DB reads). |
|
||||
| `OMNIROUTE_RTK_TRUST_PROJECT_FILTERS` | `0` | `open-sse/services/compression/engines/rtk/filterLoader.ts` | Trust user-managed RTK project filter rules without strict signature checks. |
|
||||
| `OMNI_COMPRESSION_WORKERS` | `2` | `open-sse/services/compression/compressionWorkerPool.ts` | Maximum concurrent synchronous RTK/Caveman workers; excess jobs wait FIFO. |
|
||||
| `OMNI_COMPRESSION_WORKER_TIMEOUT_MS` | `120000` | `open-sse/services/compression/compressionWorkerPool.ts` | Per-job timeout in milliseconds. Timed-out workers are terminated and the request fails open unchanged. |
|
||||
| `OMNI_COMPRESSION_WORKER_IDLE_MS` | `60000` | `open-sse/services/compression/compressionWorkerPool.ts` | Idle lifetime in milliseconds before an unused compression worker is terminated. |
|
||||
| `COMPRESSION_PIPELINE_BREAKER_ENABLED` | `false` | `open-sse/services/compression/pipelineEngineBreaker.ts` | T02 stacked-pipeline per-engine circuit-breaker master switch. **Opt-in (default off)** — when on, an engine that throws repeatedly across requests is skipped (fail-open) for a cooldown; off = byte-identical legacy behavior. |
|
||||
| `COMPRESSION_PIPELINE_BREAKER_THRESHOLD` | `3` | `open-sse/services/compression/pipelineEngineBreaker.ts` | Consecutive cross-request failures before an engine's breaker opens. |
|
||||
| `COMPRESSION_PIPELINE_BREAKER_COOLDOWN_MS` | `30000` | `open-sse/services/compression/pipelineEngineBreaker.ts` | Milliseconds an opened engine stays skipped before a half-open probe. |
|
||||
@@ -1041,6 +1044,7 @@ desktop install.
|
||||
| `EMBED_WS_PROXY_PORT` | `20131` | `src/lib/services/embedWsProxy.ts` | Port for the embedded-service WebSocket proxy server. |
|
||||
| `CLIPROXYAPI_HOST` | `127.0.0.1` | `open-sse/executors/cliproxyapi.ts` | CLIProxyAPI bridge host (legacy integration). |
|
||||
| `CLIPROXYAPI_PORT` | `5544` | `open-sse/executors/cliproxyapi.ts` | CLIProxyAPI bridge port. |
|
||||
| `CLIPROXYAPI_MANAGEMENT_KEY` | _(empty)_ | `src/lib/services/cliproxyAccountHealth.ts` | Management key for account-health reads from an externally managed CLIProxyAPI instance. |
|
||||
| `CLIPROXYAPI_CONFIG_DIR` | `~/.cli-proxy-api` | `src/lib/versionManager/processManager.ts` | CLIProxyAPI config directory. |
|
||||
| `MUX_SERVICE_PORT` | `8322` | `src/lib/services/bootstrap.ts` | Override the port where the embedded Mux (coder/mux) agent-orchestration daemon listens (always 127.0.0.1). |
|
||||
| `DARIO_HOST` | `127.0.0.1` | `open-sse/executors/dario.ts` | Dario embedded-service bind/connect host (loopback only by default). |
|
||||
@@ -1277,7 +1281,6 @@ Provider quota endpoints, network tunnels (Tailscale, Ngrok, MITM debug proxy),
|
||||
| `OMNIROUTE_SKIP_DNS_WRITE` | _(unset)_ | `src/mitm/dns/dnsConfig.ts` | Set `1` to skip writing to the hosts file when adding/removing DNS entries — for sandboxed or read-only test environments. |
|
||||
| `OMNIROUTE_SKIP_SYSTEM_TRUST` | `0` | `src/mitm/cert/install.ts`, `src/mitm/tproxy/caTrust.ts` | Test/CI-only guard: set `1` to make cert trust install/uninstall a no-op so the suite never mutates the OS trust store. Set automatically by the test setup and CI workflows. |
|
||||
| `CHANGELOG_BASE_REF` | _(auto)_ | `scripts/check/check-changelog-integrity.mjs` | Explicit base ref for the anti CHANGELOG-eat gate (defaults to the PR base branch in CI, or the highest `release/v*`). |
|
||||
| `ALLOW_CHANGELOG_REMOVALS` | `0` | `scripts/check/check-changelog-integrity.mjs` | Set `1` to turn intentional CHANGELOG bullet removals into a report instead of a failure (justify in the PR body). |
|
||||
| `ONEPROXY_ENABLED` | `true` | `src/lib/oneproxySync.ts` | Enable the 1Proxy egress pool sync. |
|
||||
| `ONEPROXY_API_URL` | `https://1proxy-api.aitradepulse.com` | `src/lib/oneproxySync.ts` | 1Proxy service API URL override. |
|
||||
| `ONEPROXY_MAX_PROXIES` | `500` | `src/lib/oneproxySync.ts` | Maximum proxies imported per sync. |
|
||||
|
||||
@@ -183,30 +183,31 @@ See [#7992](https://github.com/diegosouzapw/OmniRoute/issues/7992) and [#7111](h
|
||||
|
||||
## How It Works (Persisted Auto-Combos)
|
||||
|
||||
The Auto-Combo Engine dynamically selects the best provider/model for each request using a **14-factor scoring function** (defined in `open-sse/services/autoCombo/scoring.ts` → `DEFAULT_WEIGHTS`). Weights form a normalized distribution (custom weights are renormalized by `normalizeScoringWeights()`).
|
||||
The Auto-Combo Engine dynamically selects the best provider/model for each request using a **15-factor scoring function** (defined in `open-sse/services/autoCombo/scoring.ts` → `DEFAULT_WEIGHTS`). The default weights sum to `1.0`; custom weights are renormalized by `normalizeScoringWeights()`.
|
||||
|
||||

|
||||

|
||||
|
||||
> Source: [diagrams/auto-combo-12factor.mmd](../diagrams/auto-combo-12factor.mmd) (regenerate via `npm run docs:render-diagrams`). The filename predates the current factor set; the diagram shows 13 of the 14 factors (missing `sessionAvailability`).
|
||||
> Source: [diagrams/auto-combo-12factor.mmd](../diagrams/auto-combo-12factor.mmd) (regenerate via `npm run docs:render-diagrams`). The filename is historical; the source and rendered diagram show all 15 factors declared in `DEFAULT_WEIGHTS`.
|
||||
|
||||
| Factor | Default Weight | Description |
|
||||
| :-------------------- | :------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `health` | 0.20 | Health score from circuit breaker (CLOSED=1.0, HALF_OPEN=0.5, OPEN=0.0) |
|
||||
| `quota` | 0.15 | Remaining quota / rate-limit headroom [0..1] |
|
||||
| `costInv` | 0.15 | Inverse **blended** cost (60% input + 40% output token price, normalized) — cheaper = higher score |
|
||||
| `latencyInv` | 0.12 | Inverse p95 latency normalized to pool — faster = higher score |
|
||||
| `taskFit` | 0.08 | Task-type fitness (coding, review, planning, analysis, debugging, docs) |
|
||||
| `stability` | 0.05 | Variance-based stability (low latency stdDev / error rate) |
|
||||
| `tierPriority` | 0.05 | Account-tier priority — Ultra=1.0, Pro=0.67, Standard=0.33, Free=0.0 |
|
||||
| `tierAffinity` | 0.05 | Affinity between the candidate's tier and the manifest-recommended tier |
|
||||
| `specificityMatch` | 0.05 | Match between request specificity (manifest hint) and model tier |
|
||||
| `contextAffinity` | 0.05 | Affinity between the request's context-window need and the model's context window |
|
||||
| `sessionAvailability` | 0.05 | OAuth session availability of the candidate connection for this session (`getOAuthSessionAvailability()`; non-OAuth connections score 1.0) |
|
||||
| `connectionDensity` | 0.05 | Spreads load across connections of the same provider (anti-concentration) |
|
||||
| `quota` | 0.1429 | Remaining quota / rate-limit headroom [0..1] |
|
||||
| `health` | 0.1605 | Health score from circuit breaker (CLOSED=1.0, HALF_OPEN=0.5, OPEN=0.0) |
|
||||
| `costInv` | 0.1429 | Inverse **blended** cost (60% input + 40% output token price, normalized) — cheaper = higher score |
|
||||
| `latencyInv` | 0.1143 | Inverse p95 latency normalized to pool — faster = higher score |
|
||||
| `taskFit` | 0.0762 | Task-type fitness (coding, review, planning, analysis, debugging, docs) |
|
||||
| `stability` | 0.0476 | Variance-based stability (low latency stdDev / error rate) |
|
||||
| `tierPriority` | 0.0476 | Account-tier priority — Ultra=1.0, Pro=0.67, Standard=0.33, Free=0.0 |
|
||||
| `tierAffinity` | 0.0476 | Affinity between the candidate's tier and the manifest-recommended tier |
|
||||
| `specificityMatch` | 0.0476 | Match between request specificity (manifest hint) and model tier |
|
||||
| `contextAffinity` | 0.0476 | Affinity between the request's context-window need and the model's context window |
|
||||
| `sessionAvailability` | 0.0476 | OAuth session availability of the candidate connection for this session (`getOAuthSessionAvailability()`; non-OAuth connections score 1.0) |
|
||||
| `connectionDensity` | 0.0476 | Spreads load across connections of the same provider (anti-concentration) |
|
||||
| `cacheAffinity` | 0.00 | Rendezvous-hash affinity toward the connection likeliest to already hold this request's prompt-cache prefix (`open-sse/services/combo/promptCacheAffinity.ts`); disabled by default (#8008) |
|
||||
| `resetWindowAffinity` | 0.00 | Bias toward connections whose quota reset window is favorable (disabled by default) |
|
||||
| `quality` | 0.03 | Feedback-driven output-quality signal from the routing-event quality tracker; candidates without observations receive a neutral 0.5 |
|
||||
|
||||
**Sum:** `0.20 + 0.15 + 0.15 + 0.12 + 0.08 + 0.05 + 0.05 + 0.05 + 0.05 + 0.05 + 0.05 + 0.05 + 0.00 + 0.00 = 1.05` as literally declared in `DEFAULT_WEIGHTS`; user-configured weights are renormalized into a distribution by `normalizeScoringWeights()` before scoring.
|
||||
**Sum:** `0.1429 + 0.1605 + 0.1429 + 0.1143 + 0.0762 + (7 × 0.0476) + 0.00 + 0.00 + 0.03 = 1.0` as declared in `DEFAULT_WEIGHTS`; user-configured weights are renormalized into a distribution by `normalizeScoringWeights()` before scoring.
|
||||
|
||||
## Mode Packs
|
||||
|
||||
@@ -677,8 +678,8 @@ Including the bare `auto` (default) plus the 6 `AutoVariant` values declared in
|
||||
|
||||
## How tiers fit Auto-Combo
|
||||
|
||||
The 14-factor scoring function (`open-sse/services/autoCombo/scoring.ts`) treats tier
|
||||
membership as two signals: `tierPriority` (0.05) and `tierAffinity` (0.05). See the
|
||||
The 15-factor scoring function (`open-sse/services/autoCombo/scoring.ts`) treats tier
|
||||
membership as two signals: `tierPriority` (0.0476) and `tierAffinity` (0.0476). See the
|
||||
canonical [scoring factor table](#how-it-works-persisted-auto-combos) above for the full
|
||||
`DEFAULT_WEIGHTS` set — the per-pack overrides (ship-fast/cost-saver/quality-first/
|
||||
offline-friendly) are listed in the "Weight profiles per pack" table.
|
||||
|
||||
@@ -1,77 +1,80 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 900 566" role="img" aria-label="OmniRoute free-tier dashboard preview: about 1.53 billion documented recurring tokens per month, about 2.15 billion in the first month, 43 provider pools and 522 model budget entries. The chart shows the 19 quantified recurring pools; one-time signup credits total about 626 million and include a 10 million LongCat grant that requires KYC. Uncapped providers remain subject to rate, concurrency, account, regional, and policy limits." font-family="-apple-system,Segoe UI,Roboto,Helvetica,Arial,sans-serif">
|
||||
<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 900 566" role="img" aria-label="OmniRoute free-tier dashboard preview: about 1.51 billion documented recurring tokens per month and about 2.13 billion in the first month. The audited catalog has 40 recurring pool keys and 455 entries, 448 active and 7 discontinued; the chart represents the 20 pools with a published positive monthly token budget. One-time signup credits total about 626 million and include a 10 million LongCat grant that requires KYC. Uncapped providers remain subject to rate, concurrency, account, regional, and policy limits." font-family="-apple-system,Segoe UI,Roboto,Helvetica,Arial,sans-serif">
|
||||
<desc>Static dashboard preview of recurring token pools, first-month signup grants, and uncapped but rate-limited free-access providers.</desc>
|
||||
<rect width="900" height="566" rx="16" fill="#0d1117"/>
|
||||
<rect x="16" y="16" width="868" height="550" rx="13" fill="#161b22" stroke="#30363d"/>
|
||||
<text x="868" y="558" fill="#484f58" font-size="10.5" text-anchor="end">OmniRoute · /dashboard/free-tiers · preview mockup</text>
|
||||
<text x="32" y="50" fill="#e6edf3" font-size="18" font-weight="700">Monthly free-token budget</text>
|
||||
<text x="868" y="50" fill="#7d8590" font-size="13" text-anchor="end">43 provider pools · 522 model entries · one endpoint</text>
|
||||
<text x="868" y="50" fill="#7d8590" font-size="13" text-anchor="end">40 recurring pools · 455 catalog entries · one endpoint</text>
|
||||
<text x="32" y="84" fill="#7d8590" font-size="11.5">Steady / month</text>
|
||||
<text x="32" y="114" fill="#e6edf3" font-size="27" font-weight="800">~1.53B</text>
|
||||
<text x="32" y="114" fill="#e6edf3" font-size="27" font-weight="800">~1.51B</text>
|
||||
<text x="330" y="84" fill="#7d8590" font-size="11.5">First month (+ signup credits)</text>
|
||||
<text x="330" y="114" fill="#3fb950" font-size="27" font-weight="800">~2.15B</text>
|
||||
<text x="330" y="114" fill="#3fb950" font-size="27" font-weight="800">~2.13B</text>
|
||||
<text x="700" y="84" fill="#7d8590" font-size="11.5">ToS-flagged (you decide)</text>
|
||||
<text x="700" y="114" fill="#d29922" font-size="27" font-weight="800">15 providers</text>
|
||||
<clipPath id="bar"><rect x="32" y="132" width="836" height="16" rx="8"/></clipPath>
|
||||
<g clip-path="url(#bar)"><rect x="32" y="132" width="836" height="16" fill="#21262d"/>
|
||||
<rect x="32.0" y="132" width="512.7" height="16" fill="#6c5ce7"/>
|
||||
<rect x="545.5" y="132" width="82.6" height="16" fill="#00b894"/>
|
||||
<rect x="628.9" y="132" width="37.1" height="16" fill="#0984e3"/>
|
||||
<rect x="666.8" y="132" width="21.9" height="16" fill="#e17055"/>
|
||||
<rect x="689.5" y="132" width="21.9" height="16" fill="#fdcb6e"/>
|
||||
<rect x="712.2" y="132" width="18.9" height="16" fill="#e84393"/>
|
||||
<rect x="731.9" y="132" width="16.9" height="16" fill="#00cec9"/>
|
||||
<rect x="749.6" y="132" width="15.9" height="16" fill="#d63031"/>
|
||||
<rect x="766.3" y="132" width="14.3" height="16" fill="#a29bfe"/>
|
||||
<rect x="781.4" y="132" width="10.3" height="16" fill="#55efc4"/>
|
||||
<rect x="792.5" y="132" width="9.8" height="16" fill="#74b9ff"/>
|
||||
<rect x="803.1" y="132" width="9.3" height="16" fill="#ffeaa7"/>
|
||||
<rect x="813.2" y="132" width="8.7" height="16" fill="#fab1a0"/>
|
||||
<rect x="822.7" y="132" width="7.3" height="16" fill="#81ecec"/>
|
||||
<rect x="830.8" y="132" width="7.1" height="16" fill="#6c5ce7"/>
|
||||
<rect x="838.7" y="132" width="7.0" height="16" fill="#00b894"/>
|
||||
<rect x="846.5" y="132" width="7.0" height="16" fill="#0984e3"/>
|
||||
<rect x="854.3" y="132" width="6.8" height="16" fill="#e17055"/>
|
||||
<rect x="861.9" y="132" width="6.1" height="16" fill="#fdcb6e"/>
|
||||
<rect x="32.0" y="132" width="510.4" height="16" fill="#6c5ce7"/>
|
||||
<rect x="543.4" y="132" width="76.6" height="16" fill="#00b894"/>
|
||||
<rect x="620.9" y="132" width="76.6" height="16" fill="#0984e3"/>
|
||||
<rect x="698.5" y="132" width="30.6" height="16" fill="#e17055"/>
|
||||
<rect x="730.1" y="132" width="15.3" height="16" fill="#fdcb6e"/>
|
||||
<rect x="746.4" y="132" width="15.3" height="16" fill="#e84393"/>
|
||||
<rect x="762.7" y="132" width="12.2" height="16" fill="#00cec9"/>
|
||||
<rect x="776.0" y="132" width="10.2" height="16" fill="#d63031"/>
|
||||
<rect x="787.2" y="132" width="7.7" height="16" fill="#a29bfe"/>
|
||||
<rect x="795.8" y="132" width="5.7" height="16" fill="#55efc4"/>
|
||||
<rect x="802.5" y="132" width="5.7" height="16" fill="#74b9ff"/>
|
||||
<rect x="809.1" y="132" width="5.7" height="16" fill="#ffeaa7"/>
|
||||
<rect x="815.8" y="132" width="5.7" height="16" fill="#fab1a0"/>
|
||||
<rect x="822.4" y="132" width="5.7" height="16" fill="#81ecec"/>
|
||||
<rect x="829.1" y="132" width="5.7" height="16" fill="#6c5ce7"/>
|
||||
<rect x="835.7" y="132" width="5.7" height="16" fill="#00b894"/>
|
||||
<rect x="842.4" y="132" width="5.7" height="16" fill="#0984e3"/>
|
||||
<rect x="849.0" y="132" width="5.7" height="16" fill="#e17055"/>
|
||||
<rect x="855.7" y="132" width="5.7" height="16" fill="#fdcb6e"/>
|
||||
<rect x="862.3" y="132" width="5.7" height="16" fill="#e84393"/>
|
||||
</g>
|
||||
<text x="32" y="172" fill="#7d8590" font-size="12">Each segment = one of 19 quantified recurring pools · 43 total pools / 522 entries in the audited catalog.</text>
|
||||
<text x="32" y="172" fill="#7d8590" font-size="12">Each segment = one of 20 quantified recurring pools · 40 pools / 455 entries in the audited catalog.</text>
|
||||
<circle cx="37" cy="196" r="5" fill="#6c5ce7"/>
|
||||
<text x="48" y="200" fill="#c9d1d9" font-size="12.5">Mistral Large 3 <tspan fill="#7d8590">1.00B</tspan></text>
|
||||
<text x="48" y="200" fill="#c9d1d9" font-size="12.5">Mistral <tspan fill="#7d8590">1.00B</tspan></text>
|
||||
<circle cx="250" cy="196" r="5" fill="#00b894"/>
|
||||
<text x="261" y="200" fill="#c9d1d9" font-size="12.5">GPT-4o mini <tspan fill="#7d8590">150M</tspan></text>
|
||||
<text x="261" y="200" fill="#c9d1d9" font-size="12.5">LLM7 <tspan fill="#7d8590">150M</tspan></text>
|
||||
<circle cx="463" cy="196" r="5" fill="#0984e3"/>
|
||||
<text x="474" y="200" fill="#c9d1d9" font-size="12.5">Gemini 2.5 Flash <tspan fill="#7d8590">60M</tspan></text>
|
||||
<text x="474" y="200" fill="#c9d1d9" font-size="12.5">Nara <tspan fill="#7d8590">150M</tspan></text>
|
||||
<circle cx="676" cy="196" r="5" fill="#e17055"/>
|
||||
<text x="687" y="200" fill="#c9d1d9" font-size="12.5">GLM 4.7 <tspan fill="#7d8590">30M</tspan></text>
|
||||
<text x="687" y="200" fill="#c9d1d9" font-size="12.5">Gemini <tspan fill="#7d8590">60M</tspan></text>
|
||||
<circle cx="37" cy="226" r="5" fill="#fdcb6e"/>
|
||||
<text x="48" y="230" fill="#c9d1d9" font-size="12.5">Llama 3.3 70B <tspan fill="#7d8590">30M</tspan></text>
|
||||
<text x="48" y="230" fill="#c9d1d9" font-size="12.5">Cerebras <tspan fill="#7d8590">30M</tspan></text>
|
||||
<circle cx="250" cy="226" r="5" fill="#e84393"/>
|
||||
<text x="261" y="230" fill="#c9d1d9" font-size="12.5">Grok-3 <tspan fill="#7d8590">24M</tspan></text>
|
||||
<text x="261" y="230" fill="#c9d1d9" font-size="12.5">Cloudflare AI <tspan fill="#7d8590">30M</tspan></text>
|
||||
<circle cx="463" cy="226" r="5" fill="#00cec9"/>
|
||||
<text x="474" y="230" fill="#c9d1d9" font-size="12.5">DeepSeek V4 Pro <tspan fill="#7d8590">20M</tspan></text>
|
||||
<text x="474" y="230" fill="#c9d1d9" font-size="12.5">API Airforce <tspan fill="#7d8590">24M</tspan></text>
|
||||
<circle cx="676" cy="226" r="5" fill="#d63031"/>
|
||||
<text x="687" y="230" fill="#c9d1d9" font-size="12.5">GPT-4.1 <tspan fill="#7d8590">18M</tspan></text>
|
||||
<text x="687" y="230" fill="#c9d1d9" font-size="12.5">Ollama Cloud <tspan fill="#7d8590">20M</tspan></text>
|
||||
<circle cx="37" cy="256" r="5" fill="#a29bfe"/>
|
||||
<text x="48" y="260" fill="#c9d1d9" font-size="12.5">Llama 4 Scout <tspan fill="#7d8590">15M</tspan></text>
|
||||
<text x="48" y="260" fill="#c9d1d9" font-size="12.5">Groq <tspan fill="#7d8590">15M</tspan></text>
|
||||
<circle cx="250" cy="256" r="5" fill="#55efc4"/>
|
||||
<text x="261" y="260" fill="#c9d1d9" font-size="12.5">GPT-4o <tspan fill="#7d8590">7M</tspan></text>
|
||||
<text x="261" y="260" fill="#c9d1d9" font-size="12.5">Bluesminds <tspan fill="#7d8590">7.2M</tspan></text>
|
||||
<circle cx="463" cy="256" r="5" fill="#74b9ff"/>
|
||||
<text x="474" y="260" fill="#c9d1d9" font-size="12.5">MiniMax-M2.7 <tspan fill="#7d8590">6M</tspan></text>
|
||||
<text x="474" y="260" fill="#c9d1d9" font-size="12.5">SambaNova <tspan fill="#7d8590">6M</tspan></text>
|
||||
<circle cx="676" cy="256" r="5" fill="#ffeaa7"/>
|
||||
<text x="687" y="260" fill="#c9d1d9" font-size="12.5">Arcee Trinity Large Prev <tspan fill="#7d8590">5M</tspan></text>
|
||||
<text x="687" y="260" fill="#c9d1d9" font-size="12.5">Arcee <tspan fill="#7d8590">4.8M</tspan></text>
|
||||
<circle cx="37" cy="286" r="5" fill="#fab1a0"/>
|
||||
<text x="48" y="290" fill="#c9d1d9" font-size="12.5">Auto Free <tspan fill="#7d8590">4M</tspan></text>
|
||||
<text x="48" y="290" fill="#c9d1d9" font-size="12.5">Navy <tspan fill="#7d8590">4.5M</tspan></text>
|
||||
<circle cx="250" cy="286" r="5" fill="#81ecec"/>
|
||||
<text x="261" y="290" fill="#c9d1d9" font-size="12.5">Auto <tspan fill="#7d8590">1M</tspan></text>
|
||||
<text x="261" y="290" fill="#c9d1d9" font-size="12.5">BazaarLink <tspan fill="#7d8590">3.6M</tspan></text>
|
||||
<circle cx="463" cy="286" r="5" fill="#6c5ce7"/>
|
||||
<text x="474" y="290" fill="#c9d1d9" font-size="12.5">Command A Reasoning <tspan fill="#7d8590">800K</tspan></text>
|
||||
<text x="474" y="290" fill="#c9d1d9" font-size="12.5">OpenRouter <tspan fill="#7d8590">1.2M</tspan></text>
|
||||
<circle cx="676" cy="286" r="5" fill="#00b894"/>
|
||||
<text x="687" y="290" fill="#c9d1d9" font-size="12.5">ERNIE 4.5 VL 424B <tspan fill="#7d8590">500K</tspan></text>
|
||||
<text x="687" y="290" fill="#c9d1d9" font-size="12.5">Cohere <tspan fill="#7d8590">800K</tspan></text>
|
||||
<circle cx="37" cy="316" r="5" fill="#0984e3"/>
|
||||
<text x="48" y="320" fill="#c9d1d9" font-size="12.5">morph-v3-large <tspan fill="#7d8590">400K</tspan></text>
|
||||
<text x="48" y="320" fill="#c9d1d9" font-size="12.5">HuggingChat <tspan fill="#7d8590">500K</tspan></text>
|
||||
<circle cx="250" cy="316" r="5" fill="#e17055"/>
|
||||
<text x="261" y="320" fill="#c9d1d9" font-size="12.5">Llama 3.1 8B <tspan fill="#7d8590">200K</tspan></text>
|
||||
<text x="261" y="320" fill="#c9d1d9" font-size="12.5">Morph <tspan fill="#7d8590">400K</tspan></text>
|
||||
<circle cx="463" cy="316" r="5" fill="#fdcb6e"/>
|
||||
<text x="474" y="320" fill="#c9d1d9" font-size="12.5">Claude Sonnet 4.5 <tspan fill="#7d8590">25K</tspan></text>
|
||||
<text x="474" y="320" fill="#c9d1d9" font-size="12.5">Hugging Face <tspan fill="#7d8590">200K</tspan></text>
|
||||
<circle cx="676" cy="316" r="5" fill="#e84393"/>
|
||||
<text x="687" y="320" fill="#c9d1d9" font-size="12.5">Kiro <tspan fill="#7d8590">25K</tspan></text>
|
||||
<line x1="32" y1="386" x2="868" y2="386" stroke="#30363d"/>
|
||||
<text x="32" y="412" fill="#3fb950" font-size="13" font-weight="700">+ First month: one-time signup credits (~626M)</text>
|
||||
<rect x="32" y="421" width="90" height="22" rx="11" fill="#13311f" stroke="#238636"/>
|
||||
@@ -98,5 +101,5 @@
|
||||
<text x="299" y="466" fill="#7ee787" font-size="11.5" text-anchor="middle">nscale 5M</text>
|
||||
<rect x="32" y="492" width="836" height="34" rx="8" fill="#1c2230" stroke="#30363d"/>
|
||||
<text x="46" y="506" fill="#7d8590" font-size="12">Pool-deduped, honest counting — no inflated rate-limit ceilings. Some terms suggest personal-use only; we flag them so you decide.</text>
|
||||
<text x="46" y="520" fill="#7d8590" font-size="11.5">+ 13 recurring uncapped* providers (rate/concurrency-limited) · OpenRouter $10 → +24M/mo.</text>
|
||||
<text x="46" y="520" fill="#7d8590" font-size="11.5">+ 14 recurring uncapped* providers (rate/concurrency-limited) · OpenRouter $10 → +24M/mo.</text>
|
||||
</svg>
|
||||
|
||||
|
Before Width: | Height: | Size: 8.7 KiB After Width: | Height: | Size: 8.9 KiB |
@@ -1,13 +1,13 @@
|
||||
---
|
||||
title: "Guardrails"
|
||||
version: 3.8.50
|
||||
lastUpdated: 2026-08-14
|
||||
lastUpdated: 2026-08-24
|
||||
---
|
||||
|
||||
# Guardrails
|
||||
|
||||
> **Source of truth:** `src/lib/guardrails/`
|
||||
> **Last updated:** 2026-08-15 — v3.8.50 (Video Bridge broker confinement)
|
||||
> **Last updated:** 2026-08-24 — v3.8.50 (Video Bridge visual dedup hardening + focused captions)
|
||||
|
||||
Guardrails enforce safety, policy, and content transformations at the boundary
|
||||
between OmniRoute and upstream providers. Each guardrail can inspect (and
|
||||
@@ -327,30 +327,106 @@ fixed FFmpeg pass over the already validated local stream, select bounded
|
||||
`showinfo` scene timestamps, and fall back deterministically to the same
|
||||
uniform midpoints on detector failure, timeout, malformed output, or an empty
|
||||
candidate set. Segment-aware mode allocates midpoint samples proportionally to
|
||||
the validated scene intervals. The hard 16-frame cap is
|
||||
applied after selection in every policy. A caller may optionally provide a
|
||||
the validated scene intervals; segment-aware evidence and fallback behavior are
|
||||
detailed below. The hard 16-frame cap is
|
||||
applied after selection in every policy. When a scene-aware request has only a
|
||||
one-frame budget, it uses the uniform midpoint of the active full-video or focus
|
||||
window and reports `policyEffective: uniform`: a single selected scene frame
|
||||
cannot preserve both temporal ends. A caller may optionally provide a
|
||||
finite focus window (`start`/`end` seconds); bounds are clamped to the media
|
||||
duration, reversed or non-finite windows are rejected, and all sampling
|
||||
policies are performed only inside the normalized interval. The resulting
|
||||
window is included in sampling metadata and in the untrusted description
|
||||
prefix so downstream models can distinguish a focused excerpt from the full
|
||||
timeline.
|
||||
|
||||
Semantic caption focus is a separate, explicit setting. The default `full`
|
||||
analysis mode preserves the existing frame prompt and never forwards request
|
||||
text to the caption model. In `focused` mode, the bridge reads only the latest
|
||||
non-empty user-authored `text`/`input_text` from the same Chat or Responses
|
||||
container, normalizes it to NFC, collapses control characters and whitespace,
|
||||
and limits it to 500 Unicode code points. An empty result falls back to the
|
||||
exact `full` prompt. A usable hint is serialized as JSON in a dedicated
|
||||
untrusted-user-context block and may only prioritize observable details; it
|
||||
cannot override the separate warning against following instructions visible
|
||||
or audible in the media. Textual focus never infers `start`/`end` or changes
|
||||
the temporal sampler.
|
||||
|
||||
#### FU-07 structural segment evidence
|
||||
|
||||
`segment_aware` uses one bounded pre-analysis pass over the already validated
|
||||
local video stream. The fixed filter chain first scales to at most 320 pixels
|
||||
wide, detects scene changes and frozen intervals, then samples at 1 frame per
|
||||
second for blur, average luma, and spatial/temporal information. The pass is
|
||||
limited to 600 structural samples, one FFmpeg/filter thread, the same
|
||||
`file`-only protocol and container allowlists, a 1 MiB process-output bound,
|
||||
and at most 30 seconds inside the broker's shared abort/deadline. It never
|
||||
accepts a command, filter, path, or URL from the request.
|
||||
|
||||
The structural values are deterministic sampling evidence, not semantic video
|
||||
understanding. They do not infer subjects, actions, captions, speech, or user
|
||||
intent. Scene and freeze boundaries form segments; freeze coverage, blur,
|
||||
exposure, spatial detail, and temporal change only influence how the existing
|
||||
1–16 frame budget is allocated. A fully frozen segment is capped at one frame,
|
||||
while non-frozen segments compete for the remaining budget. When boundaries
|
||||
outnumber frames, uniform timeline coverage is retained so rapid early cuts
|
||||
cannot hide a long trailing segment. Scene boundaries within the 1-second
|
||||
analysis resolution of a freeze boundary are coalesced.
|
||||
|
||||
Missing filters, malformed/empty evidence, a detector error, or the bounded
|
||||
pre-analysis timeout fail open to the exact uniform midpoint policy. A caller
|
||||
abort or broker deadline does not fail open: it terminates the in-flight
|
||||
subprocess, prevents later frame extraction, and the private temporary tree is
|
||||
removed in `finally`.
|
||||
|
||||
`scripts/perf/video-bridge-fu07-eval.ts` generates deterministic real FFmpeg
|
||||
fixtures for post-dedup caption-call savings, dense-motion budget allocation,
|
||||
blur/exposure/SI-TI evidence, rapid cuts with a long tail, and gradual-fade
|
||||
false positives. It records pre-analysis wall time and, where `/usr/bin/time`
|
||||
is available, child CPU and peak RSS. Its quality checks are structural oracles
|
||||
only. Real caption-model quality remains `HOLD` because this harness has no
|
||||
authorized endpoint or frozen judge. Monetary savings also remain `HOLD`
|
||||
unless `--caption-cost-per-call-usd` supplies an explicit positive per-call
|
||||
estimate; the script never fabricates either result.
|
||||
|
||||
Each frame is limited to 4 MiB, all raw frames together to 23 MiB, and the
|
||||
serialized broker response to 32 MiB. A private temporary directory is removed
|
||||
in `finally`. OmniRoute does not bundle FFmpeg and does not accept a custom
|
||||
executable path. Before captioning, the bridge applies a conservative visual
|
||||
deduplication pass: each JPEG is reduced to a 16×16 grayscale buffer and is
|
||||
compared only with the last frame retained, using a fixed similarity threshold
|
||||
of 0.04 — a deliberate constant chosen for predictability, not a runtime
|
||||
setting. The first and final timeline frames
|
||||
are always retained; comparator or decoder errors fail open and keep coverage.
|
||||
The output metadata reports how many frames were dropped.
|
||||
compared only with the last frame retained. For a requested caption budget
|
||||
above one frame, extraction supplies a
|
||||
bounded candidate pool of up to twice that budget and never more than 16 frames.
|
||||
The requested cap is applied only after deduplication, with the first and final
|
||||
selected candidates preserved during final thinning when the budget is at least
|
||||
two. The versioned
|
||||
`grayscale-16x16-mean-cells-v2` policy uses the larger of mean luma delta and
|
||||
the ratio of thumbnail cells whose normalized delta is at least 0.05. The
|
||||
duplicate threshold is the constant 0.04, chosen for predictability rather than
|
||||
exposed as a runtime setting. This secondary
|
||||
high-contrast signal preserves small motion and visible-text changes that a
|
||||
mean-only comparison can hide. Comparator or decoder errors fail open and keep
|
||||
coverage. Output metadata separates extracted candidates, successfully used
|
||||
frames, and visual duplicates dropped.
|
||||
|
||||
An explicitly marked video part may request a timestamped contact sheet. The
|
||||
bridge builds at most a 4-column, 16-frame JPEG grid and labels the resulting
|
||||
observation with every source timestamp. If `sharp` cannot decode or compose
|
||||
the grid, the bridge falls back to the individual JPEG frames; a client abort
|
||||
still propagates through the sheet operation.
|
||||
bridge builds at most a 4-column, 16-frame JPEG grid. Every 512-pixel cell burns
|
||||
its source timestamp into a high-contrast bottom band, while the same timestamps
|
||||
remain in textual metadata for downstream association and audit. The complete
|
||||
JPEG remains capped at 32 MiB. If `sharp` cannot decode or compose the grid, the
|
||||
bridge falls back to the individual JPEG frames; a client abort still propagates
|
||||
through the sheet operation.
|
||||
|
||||
Promotion evidence is deliberately separate from the synthetic composition
|
||||
microbenchmark. `scripts/perf/video-bridge-contact-sheet-eval.ts` defines a
|
||||
schema-versioned A/B harness for real OpenAI-compatible vision models. It measures
|
||||
provider-reported tokens, end-to-end wall latency (including sheet composition),
|
||||
model-call count, and manifest-defined fact retention. Raw model responses are not
|
||||
written to the report; only SHA-256 digests and matched fact IDs are retained. The
|
||||
harness makes no network or paid model call unless `--execute-real` is passed and
|
||||
`--model`, `OMNIROUTE_BASE_URL`, and `OMNIROUTE_API_KEY` are configured. Without
|
||||
that explicit real run, its machine-readable verdict remains `HOLD`; synthetic
|
||||
payload/call-count measurements alone are not promotion evidence.
|
||||
|
||||
Callers may attach an optional `transcript.cues` array to a supported video
|
||||
part when they already possess aligned text. Each cue must carry `text`, a
|
||||
@@ -378,14 +454,39 @@ or download a second media copy; without that explicit track, it remains
|
||||
video-only.
|
||||
|
||||
The internal `/api/modality-bridge/video/drilldown` lifecycle is a separate,
|
||||
loopback/token-authenticated cache. It stores at most 16 JPEG frames per entry,
|
||||
keeps entries isolated by session and video reference, expires them after ten
|
||||
minutes, and supports bounded `start`/`end` reads or explicit session deletion.
|
||||
Besides the per-entry limits, the cache enforces a global 256 MiB decoded-byte
|
||||
budget: least-recently-used entries are evicted until new content fits, and an
|
||||
entry larger than the whole budget is rejected outright.
|
||||
It only slices materialized frames and cannot increase the cost of the primary
|
||||
video request.
|
||||
loopback/token-authenticated cache substrate. Every operation also requires a
|
||||
canonical opaque principal ID. Before a production caller is enabled, it must
|
||||
derive that ID from the authenticated tenant and must never forward a
|
||||
client-selected value. Cache keys bind that principal to canonical session and
|
||||
video-reference IDs, store only their SHA-256-derived keys, and scope both reads
|
||||
and deletion to the same principal. The cache stores at most 16 derived JPEG
|
||||
frames per entry, expires them after ten minutes, and supports bounded
|
||||
`start`/`end` reads or explicit session deletion.
|
||||
|
||||
Each principal is limited to 16 entries and 64 MiB of canonical JPEG data. Those
|
||||
limits are independent from the global 64-entry/256 MiB ceiling: principal quota
|
||||
pressure evicts only that principal's least-recently-used entries before global
|
||||
LRU eviction is considered. Expired entries are swept from both principal and
|
||||
global accounting on cache activity, while cancellation and validation failure do
|
||||
not commit a partial replacement.
|
||||
|
||||
The cache rejects non-canonical Base64, excess padding, non-JPEG media, malformed or
|
||||
truncated JPEGs, and JPEGs that produce a warning during a bounded full-image `sharp`
|
||||
decode. It re-encodes each accepted image as a canonical JPEG, derives width and height
|
||||
from the decoded bytes instead of trusting caller fields, and discards any trailing
|
||||
polyglot bytes rather than retaining them. Only the bounded canonical compressed buffer
|
||||
is charged to both quotas. The JSON wire limit includes Base64 overhead for the 32 MiB
|
||||
decoded-input ceiling. Every
|
||||
stored derivation records its validated JPEG format/resolution, sampling policy,
|
||||
derivation version, creation time, server-computed content hash, and hashed parent
|
||||
reference plus the trusted caller's parent-content hash. Cancellation is checked
|
||||
between asynchronous decode/hash phases before the atomic cache commit.
|
||||
|
||||
This tranche does not yet connect a production producer to the route and does not
|
||||
provide multi-resolution variant selection. The transparent Video Bridge request
|
||||
path therefore incurs no added work, while tenant-bound principal derivation and
|
||||
the full FU-08 multi-resolution lifecycle remain explicit follow-up work rather
|
||||
than documented as complete behavior.
|
||||
|
||||
Frames are captioned sequentially with the configured Video model. An empty
|
||||
Video override inherits the Vision setting; if both are empty, the Vision
|
||||
@@ -399,9 +500,16 @@ including a fallback model; the bridge reports `mixed` when different frames
|
||||
were produced by different models. A cache hit reuses that producer identity
|
||||
instead of relabeling it as the requested routing plan. The whole-video result
|
||||
cache is keyed on every input that changes the output — prompt, effective
|
||||
model, sampling policy, frame count, focus window, `transcript`,
|
||||
model, sampling policy, frame count, semantic analysis mode, the SHA-256
|
||||
fingerprint of the normalized focus hint, focus window, `transcript`,
|
||||
`audioTranscript`, and the contact-sheet flag — so changing any of those
|
||||
dimensions is a cache miss, never a stale reuse.
|
||||
dimensions is a cache miss, never a stale reuse. The visual dedup policy
|
||||
version, threshold, and bounded candidate-frame count are also explicit in the
|
||||
result-cache key and metadata; a policy change therefore cannot reuse a stale
|
||||
whole-video description. Result-cache v4 metadata keeps the mode and
|
||||
fingerprint, never the raw user task. Guardrail metadata reports both the
|
||||
requested and effective analysis modes; a requested `focused` mode without
|
||||
usable user text is reported as effectively `full`.
|
||||
|
||||
The guardrail extracts every supported video part but describes no more than
|
||||
`modalityBridgeVideoMaxVideos`. For a target proven to have
|
||||
@@ -417,6 +525,7 @@ Runtime settings are DB-backed and Zod-validated:
|
||||
| Key | Default | Range / behavior |
|
||||
| ----------------------------------- | ----------- | --------------------------------------------------------------------------------------------------- |
|
||||
| `modalityBridgeVideoEnabled` | `false` | Optional runtime, opt-in |
|
||||
| `modalityBridgeVideoAnalysisMode` | `"full"` | `full` preserves generic captions; `focused` uses bounded, untrusted latest-user context |
|
||||
| `modalityBridgeVideoModel` | `""` | Inherit the Vision Bridge model |
|
||||
| `modalityBridgeVideoFrameCount` | `8` | 1–16 |
|
||||
| `modalityBridgeVideoSamplingPolicy` | `"uniform"` | `uniform`, `scene_aware`, or proportional `segment_aware`; detector failure falls back to `uniform` |
|
||||
@@ -659,7 +768,8 @@ Audio uses `modalityBridgeAudioEnabled`, `modalityBridgeAudioModel`,
|
||||
`modalityBridgeCache*` settings. Audio has no legacy-key fallback because these
|
||||
keys were introduced with the Modality Bridge schema.
|
||||
|
||||
Video uses `modalityBridgeVideoEnabled`, `modalityBridgeVideoModel`,
|
||||
Video uses `modalityBridgeVideoEnabled`, `modalityBridgeVideoAnalysisMode`,
|
||||
`modalityBridgeVideoModel`,
|
||||
`modalityBridgeVideoFrameCount`, `modalityBridgeVideoSamplingPolicy`,
|
||||
`modalityBridgeVideoMaxVideos`, and
|
||||
`modalityBridgeVideoTimeout`, plus the shared `modalityBridgeCache*` settings.
|
||||
|
||||
@@ -2,6 +2,7 @@ import createNextIntlPlugin from "next-intl/plugin";
|
||||
import { createMDX } from "fumadocs-mdx/next";
|
||||
import { dirname } from "node:path";
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { betterSqlite3AliasFor } from "./scripts/build/better-sqlite3-stub-flag.mjs";
|
||||
import { mitmManagerAliasFor } from "./scripts/build/mitm-stub-flag.mjs";
|
||||
import { normalizeBasePath } from "./scripts/build/normalizeBasePath.mjs";
|
||||
import {
|
||||
@@ -138,10 +139,14 @@ const nextConfig = {
|
||||
// the stub to every npm/Electron/VPS artifact and broke Agent Bridge
|
||||
// start for all non-Docker users (#6344). See scripts/build/mitm-stub-flag.mjs.
|
||||
...mitmManagerAliasFor(process.env),
|
||||
// Build-time stub so the bundler never traces the native better-sqlite3
|
||||
// addon into a build worker (SIGABRT at worker teardown). Runtime still
|
||||
// uses the real package via serverExternalPackages. (#10060)
|
||||
"better-sqlite3": "./src/lib/db/better-sqlite3.stub.js",
|
||||
// better-sqlite3 → build-time stub ONLY where the build worker actually
|
||||
// aborts while tracing the native addon (SIGABRT at worker teardown,
|
||||
// #10060); opt in with OMNIROUTE_BETTER_SQLITE3_STUB=1. The alias used to
|
||||
// be unconditional on the premise that serverExternalPackages still won
|
||||
// at runtime — it does not: resolveAlias rewrites the request before the
|
||||
// externals check, so the stub was bundled and EVERY route answered 500
|
||||
// (#11343). See scripts/build/better-sqlite3-stub-flag.mjs.
|
||||
...betterSqlite3AliasFor(process.env),
|
||||
...minimalBuildAliases,
|
||||
},
|
||||
// src/lib/agentSkills/generator.ts builds its fs base path from a runtime
|
||||
|
||||
@@ -287,6 +287,19 @@ export const AUDIO_TRANSLATION_PROVIDERS: Record<string, AudioProvider> = {
|
||||
};
|
||||
|
||||
export const AUDIO_SPEECH_PROVIDERS: Record<string, AudioProvider> = {
|
||||
google: {
|
||||
id: "google",
|
||||
credentialProviderId: "gemini",
|
||||
baseUrl: "https://generativelanguage.googleapis.com/v1beta/models",
|
||||
authType: "apikey",
|
||||
authHeader: "x-goog-api-key",
|
||||
format: "gemini-tts",
|
||||
models: [
|
||||
{ id: "gemini-3.1-flash-tts-preview", name: "Gemini 3.1 Flash TTS" },
|
||||
{ id: "gemini-2.5-flash-preview-tts", name: "Gemini 2.5 Flash TTS" },
|
||||
{ id: "gemini-2.5-pro-preview-tts", name: "Gemini 2.5 Pro TTS" },
|
||||
],
|
||||
},
|
||||
vertex: {
|
||||
id: "vertex",
|
||||
baseUrl: "https://us-central1-aiplatform.googleapis.com/v1",
|
||||
|
||||
@@ -16,7 +16,7 @@ import type { FreeModelBudget } from "./freeModelCatalog.ts";
|
||||
* rewrites file timestamps on every deploy, which would report a months-old
|
||||
* catalog as "updated today". Bump this whenever the entries below change.
|
||||
*/
|
||||
export const FREE_CATALOG_CURATED_AT = "2026-08-18";
|
||||
export const FREE_CATALOG_CURATED_AT = "2026-08-20";
|
||||
|
||||
export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [
|
||||
{ provider: "chatgpt-web", modelId: "gpt-5.6-luna-free", displayName: "GPT-5.6 Luna (Free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "chatgpt-web-free", tos: "caution" },
|
||||
@@ -318,6 +318,7 @@ export const FREE_MODEL_BUDGETS: FreeModelBudget[] = [
|
||||
{ provider: "opencode-zen", modelId: "opencode/north-mini-code-free", displayName: "North Mini Code (free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "opencode-zen-free", tos: "caution" },
|
||||
{ provider: "opencode-zen", modelId: "opencode/nemotron-3-ultra-free", displayName: "Nemotron 3 Ultra (free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-uncapped", poolKey: "opencode-zen-free", tos: "caution" },
|
||||
{ provider: "openrouter", modelId: "auto", displayName: "Auto (Best Available)", monthlyTokens: 1200000, creditTokens: 0, freeType: "recurring-daily", poolKey: "openrouter-free", tos: "caution" },
|
||||
{ provider: "openrouter", modelId: "stealth/ox-alpha", displayName: "Stealth Ox Alpha (free)", monthlyTokens: 0, creditTokens: 0, freeType: "recurring-daily", poolKey: "openrouter-free", tos: "caution" },
|
||||
{ provider: "pollinations", modelId: "openai", displayName: "OpenAI (Pollinations)", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "pollinations", tos: "caution" },
|
||||
{ provider: "pollinations", modelId: "openai-fast", displayName: "OpenAI Fast (Pollinations)", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "pollinations", tos: "caution" },
|
||||
{ provider: "pollinations", modelId: "openai-large", displayName: "OpenAI Large (Pollinations)", monthlyTokens: 0, creditTokens: 0, freeType: "keyless", poolKey: "pollinations", tos: "caution" },
|
||||
|
||||
@@ -48,6 +48,17 @@ export const GLM_SHARED_MODELS = Object.freeze([
|
||||
supportsReasoning: true,
|
||||
supportedThinkingEfforts: ["low"],
|
||||
},
|
||||
{
|
||||
// Explicit alias for the upstream default (max) — pins reasoning_effort so
|
||||
// the tier survives an upstream default change, and mirrors glm-5.2-max UX.
|
||||
id: "glm-5.3-max",
|
||||
name: "GLM 5.3 Max",
|
||||
contextLength: 1000000,
|
||||
maxOutputTokens: 131072,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
supportedThinkingEfforts: ["max"],
|
||||
},
|
||||
{
|
||||
// GLM-5.2 has two positive effective tiers: low/medium map to high and xhigh
|
||||
// maps to max; disabling thinking remains the separate thinking toggle.
|
||||
|
||||
@@ -181,6 +181,29 @@ export function getRegistryEntry(provider: string): RegistryEntry | null {
|
||||
return REGISTRY[provider] || _byAlias.get(provider) || null;
|
||||
}
|
||||
|
||||
/** Resolve only a model's explicit reasoning vocabulary. */
|
||||
export function getRegistryModelThinkingEfforts(
|
||||
provider: string,
|
||||
modelId: string
|
||||
): readonly string[] | undefined {
|
||||
const entry = getRegistryEntry(provider);
|
||||
if (!entry) return undefined;
|
||||
const model = entry.models.find((candidate) => candidate.id === modelId);
|
||||
return model?.supportedThinkingEfforts;
|
||||
}
|
||||
|
||||
/** Resolve a model's explicit reasoning vocabulary before its provider fallback. */
|
||||
export function getRegistryThinkingEfforts(
|
||||
provider: string,
|
||||
modelId: string
|
||||
): readonly string[] | undefined {
|
||||
const entry = getRegistryEntry(provider);
|
||||
if (!entry) return undefined;
|
||||
const modelEfforts = getRegistryModelThinkingEfforts(provider, modelId);
|
||||
if (modelEfforts !== undefined) return modelEfforts;
|
||||
return entry.defaultSupportedThinkingEfforts;
|
||||
}
|
||||
|
||||
/**
|
||||
* Decide whether a non-empty live catalog may exclude omitted static models
|
||||
* during request routing and wildcard expansion.
|
||||
|
||||
@@ -70,6 +70,8 @@ import { togetherProvider } from "./registry/together/index.ts";
|
||||
import { cohereProvider } from "./registry/cohere/index.ts";
|
||||
import { cursorProvider, cursor_apiProvider } from "./registry/cursor/index.ts";
|
||||
import { volcengineProvider } from "./registry/volcengine/index.ts";
|
||||
import { volcengine_agent_planProvider } from "./registry/volcengine/agent-plan/index.ts";
|
||||
import { volcengine_coding_planProvider } from "./registry/volcengine/coding-plan/index.ts";
|
||||
import { freetheaiProvider } from "./registry/freetheai/index.ts";
|
||||
import { g4f_groqProvider } from "./registry/g4f-groq/index.ts";
|
||||
import { g4f_geminiProvider } from "./registry/g4f-gemini/index.ts";
|
||||
@@ -337,6 +339,8 @@ export const REGISTRY: Record<string, RegistryEntry> = {
|
||||
cursor: cursorProvider,
|
||||
"cursor-api": cursor_apiProvider,
|
||||
volcengine: volcengineProvider,
|
||||
"volcengine-agent-plan": volcengine_agent_planProvider,
|
||||
"volcengine-coding-plan": volcengine_coding_planProvider,
|
||||
freetheai: freetheaiProvider,
|
||||
"g4f-groq": g4f_groqProvider,
|
||||
"g4f-gemini": g4f_geminiProvider,
|
||||
|
||||
@@ -9,6 +9,7 @@ export const ollama_cloudProvider: RegistryEntry = {
|
||||
modelsUrl: "https://ollama.com/api/tags",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
defaultSupportedThinkingEfforts: ["none", "low", "medium", "high", "max"],
|
||||
// Note: rate limits vary by plan (free = "Light usage", Pro = more, Max = 5x Pro).
|
||||
// Users can generate API keys at https://ollama.com/settings/keys
|
||||
models: [
|
||||
@@ -24,23 +25,20 @@ export const ollama_cloudProvider: RegistryEntry = {
|
||||
supportsReasoning: true,
|
||||
supportedThinkingEfforts: ["low", "medium", "high"],
|
||||
},
|
||||
// #10788: Ollama Cloud accepts low|medium|high|max|none uniformly across
|
||||
// its reasoning-capable models (see supportsMaxEffortForProvider's
|
||||
// isOllamaCloud comment in open-sse/executors/base/reasoningEffort.ts) —
|
||||
// declare supportedThinkingEfforts so appendSyncedEffortVariants() (which
|
||||
// runs before static-model capability enrichment) can synthesize the
|
||||
// catalog's selectable -low/-high/-max variant ids for these models.
|
||||
// #10788: these models accept none|low|medium|high|max. Keep their explicit
|
||||
// declarations aligned with the provider fallback so the static and synced
|
||||
// catalog paths expose the same native vocabulary.
|
||||
{
|
||||
id: "deepseek-v4-pro",
|
||||
name: "DeepSeek V4 Pro",
|
||||
supportsReasoning: true,
|
||||
supportedThinkingEfforts: ["low", "medium", "high", "max"],
|
||||
supportedThinkingEfforts: ["none", "low", "medium", "high", "max"],
|
||||
},
|
||||
{
|
||||
id: "deepseek-v4-flash",
|
||||
name: "DeepSeek V4 Flash",
|
||||
supportsReasoning: true,
|
||||
supportedThinkingEfforts: ["low", "medium", "high", "max"],
|
||||
supportedThinkingEfforts: ["none", "low", "medium", "high", "max"],
|
||||
},
|
||||
{ id: "kimi-k2.6", name: "Kimi K2.6" },
|
||||
// Ollama Cloud accepts low|medium|high|max|none and rejects xhigh, so the
|
||||
@@ -50,14 +48,14 @@ export const ollama_cloudProvider: RegistryEntry = {
|
||||
name: "GLM 5.1",
|
||||
supportsReasoning: true,
|
||||
supportsXHighEffort: false,
|
||||
supportedThinkingEfforts: ["low", "medium", "high", "max"],
|
||||
supportedThinkingEfforts: ["none", "low", "medium", "high", "max"],
|
||||
},
|
||||
{
|
||||
id: "glm-5.2",
|
||||
name: "GLM 5.2",
|
||||
supportsReasoning: true,
|
||||
supportsXHighEffort: false,
|
||||
supportedThinkingEfforts: ["low", "medium", "high", "max"],
|
||||
supportedThinkingEfforts: ["none", "low", "medium", "high", "max"],
|
||||
},
|
||||
// #3110: MiniMax M3 via Ollama
|
||||
{ id: "minimax-m3", name: "MiniMax M3", contextLength: 1048576, supportsVision: true },
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
import type { RegistryEntry, RegistryModel } from "../../../shared.ts";
|
||||
|
||||
/**
|
||||
* Volcano Ark Agent Plan models.
|
||||
*
|
||||
* The Agent Plan subscription (console.volcengine.com/ark/subscription/agent-plan)
|
||||
* is served by the Plan API endpoint — `/api/plan/v3` — which differs from both the
|
||||
* standard pay-per-use API (`/api/v3`) and the Coding Plan API (`/api/coding/v3`).
|
||||
* The Plan API has NO `/models` listing endpoint (returns 404); key validation falls
|
||||
* back to a chat probe against the first model. Model IDs below verified live against
|
||||
* /api/plan/v3/chat/completions (all return 200).
|
||||
*/
|
||||
export const VOLCENGINE_AGENT_PLAN_MODELS: RegistryModel[] = [
|
||||
{
|
||||
id: "doubao-seed-evolving",
|
||||
name: "Doubao Seed Evolving (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "doubao-seed-2-1-turbo-260628",
|
||||
name: "Doubao Seed 2.1 Turbo (Agent Plan)",
|
||||
contextLength: 262144,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "doubao-seed-2-0-lite-260215",
|
||||
name: "Doubao Seed 2.0 Lite (Agent Plan)",
|
||||
contextLength: 262144,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "doubao-seed-2-0-mini-260215",
|
||||
name: "Doubao Seed 2.0 Mini (Agent Plan)",
|
||||
contextLength: 262144,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "deepseek-v4-flash-ga-260731",
|
||||
name: "DeepSeek V4 Flash GA (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "kimi-k3",
|
||||
name: "Kimi K3 (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "glm-5-2-260617",
|
||||
name: "GLM 5.2 (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "kimi-k2.7-code",
|
||||
name: "Kimi K2.7 Code (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "minimax-m3",
|
||||
name: "MiniMax M3 (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "deepseek-v4-pro-260425",
|
||||
name: "DeepSeek V4 Pro (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "minimax-m2.7",
|
||||
name: "MiniMax M2.7 (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "kimi-k2.6",
|
||||
name: "Kimi K2.6 (Agent Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
];
|
||||
|
||||
export const volcengine_agent_planProvider: RegistryEntry = {
|
||||
id: "volcengine-agent-plan",
|
||||
alias: "veap",
|
||||
format: "openai",
|
||||
executor: "default",
|
||||
baseUrl: "https://ark.cn-beijing.volces.com/api/plan/v3/chat/completions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
models: VOLCENGINE_AGENT_PLAN_MODELS,
|
||||
};
|
||||
@@ -0,0 +1,92 @@
|
||||
import type { RegistryEntry, RegistryModel } from "../../../shared.ts";
|
||||
|
||||
/**
|
||||
* Volcano Ark Coding Plan models.
|
||||
*
|
||||
* The Coding Plan subscription (console.volcengine.com/ark/subscription/coding-plan)
|
||||
* is served by a DEDICATED endpoint — `/api/coding/v3` — which differs from both the
|
||||
* standard pay-per-use API (`/api/v3`) and the Agent Plan API (`/api/plan/v3`). Using
|
||||
* the wrong base URL returns HTTP 401 "The API key or AK/SK ... is missing or invalid"
|
||||
* even with a valid Coding Plan key. Model IDs below verified live against
|
||||
* /api/coding/v3/chat/completions (all return 200).
|
||||
*/
|
||||
export const VOLCENGINE_CODING_PLAN_MODELS: RegistryModel[] = [
|
||||
{
|
||||
id: "doubao-seed-2-1-turbo",
|
||||
name: "Doubao Seed 2.1 Turbo (Coding Plan)",
|
||||
contextLength: 262144,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "doubao-seed-2.0-lite",
|
||||
name: "Doubao Seed 2.0 Lite (Coding Plan)",
|
||||
contextLength: 262144,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "deepseek-v4-flash",
|
||||
name: "DeepSeek V4 Flash (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "glm-5.2",
|
||||
name: "GLM 5.2 (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "kimi-k2.7-code",
|
||||
name: "Kimi K2.7 Code (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsVision: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "minimax-m3",
|
||||
name: "MiniMax M3 (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "deepseek-v4-pro",
|
||||
name: "DeepSeek V4 Pro (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "minimax-m2.7",
|
||||
name: "MiniMax M2.7 (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
{
|
||||
id: "kimi-k2.6",
|
||||
name: "Kimi K2.6 (Coding Plan)",
|
||||
contextLength: 1048576,
|
||||
toolCalling: true,
|
||||
supportsReasoning: true,
|
||||
},
|
||||
];
|
||||
|
||||
export const volcengine_coding_planProvider: RegistryEntry = {
|
||||
id: "volcengine-coding-plan",
|
||||
alias: "vecp",
|
||||
format: "openai",
|
||||
executor: "default",
|
||||
baseUrl: "https://ark.cn-beijing.volces.com/api/coding/v3/chat/completions",
|
||||
authType: "apikey",
|
||||
authHeader: "bearer",
|
||||
models: VOLCENGINE_CODING_PLAN_MODELS,
|
||||
modelsUrl: "/models",
|
||||
};
|
||||
@@ -139,6 +139,9 @@ export interface RegistryEntry {
|
||||
requestDefaults?: ProviderRequestDefaults;
|
||||
oauth?: RegistryOAuth;
|
||||
models: RegistryModel[];
|
||||
/** Provider-native reasoning vocabulary for reasoning-capable passthrough models
|
||||
* that do not have an explicit per-model declaration. */
|
||||
defaultSupportedThinkingEfforts?: readonly string[];
|
||||
modelsUrl?: string;
|
||||
/** Prefix to prepend to model IDs before upstream API calls (e.g. "accounts/fireworks/models/") */
|
||||
modelIdPrefix?: string;
|
||||
|
||||
81
open-sse/executors/geminiTts.ts
Normal file
@@ -0,0 +1,81 @@
|
||||
import { Buffer } from "node:buffer";
|
||||
import { extractInlineAudio, parsePcmSampleRate, pcmToWav } from "./vertexMedia.ts";
|
||||
import { CORS_HEADERS } from "../utils/cors.ts";
|
||||
import { upstreamErrorResponse } from "../utils/audioResponse.ts";
|
||||
import { errorResponse } from "../utils/error.ts";
|
||||
|
||||
type GeminiTtsCredentials = {
|
||||
apiKey?: string | null;
|
||||
accessToken?: string | null;
|
||||
};
|
||||
|
||||
export class GeminiTtsUpstreamError extends Error {
|
||||
constructor(
|
||||
public readonly response: Response,
|
||||
public readonly body: string
|
||||
) {
|
||||
super(`Gemini TTS upstream error (${response.status})`);
|
||||
}
|
||||
}
|
||||
|
||||
export async function geminiGenerateSpeech(
|
||||
credentials: GeminiTtsCredentials,
|
||||
options: { model: string; text: string; voice: string }
|
||||
): Promise<Buffer> {
|
||||
const headers: Record<string, string> = { "Content-Type": "application/json" };
|
||||
if (credentials.apiKey) {
|
||||
headers["x-goog-api-key"] = credentials.apiKey;
|
||||
} else if (credentials.accessToken) {
|
||||
headers.Authorization = `Bearer ${credentials.accessToken}`;
|
||||
}
|
||||
|
||||
const response = await fetch(
|
||||
`https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent(options.model)}:generateContent`,
|
||||
{
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({
|
||||
contents: [{ parts: [{ text: options.text }] }],
|
||||
generationConfig: {
|
||||
responseModalities: ["AUDIO"],
|
||||
speechConfig: {
|
||||
voiceConfig: {
|
||||
prebuiltVoiceConfig: { voiceName: options.voice },
|
||||
},
|
||||
},
|
||||
},
|
||||
}),
|
||||
}
|
||||
);
|
||||
if (!response.ok) {
|
||||
throw new GeminiTtsUpstreamError(response, await response.text());
|
||||
}
|
||||
|
||||
const inline = extractInlineAudio(await response.json());
|
||||
if (!inline) throw new Error("Gemini TTS response did not contain audio data");
|
||||
return pcmToWav(Buffer.from(inline.base64, "base64"), parsePcmSampleRate(inline.mimeType));
|
||||
}
|
||||
|
||||
export async function handleGeminiTtsSpeech(
|
||||
credentials: GeminiTtsCredentials,
|
||||
options: { model: string; text: string; voice?: unknown }
|
||||
): Promise<Response> {
|
||||
try {
|
||||
const wav = await geminiGenerateSpeech(credentials, {
|
||||
model: options.model,
|
||||
text: options.text,
|
||||
voice:
|
||||
typeof options.voice === "string" && options.voice.trim() ? options.voice.trim() : "Kore",
|
||||
});
|
||||
return new Response(new Uint8Array(wav), {
|
||||
status: 200,
|
||||
headers: { ...CORS_HEADERS, "Content-Type": "audio/wav" },
|
||||
});
|
||||
} catch (error) {
|
||||
if (error instanceof GeminiTtsUpstreamError) {
|
||||
return upstreamErrorResponse(error.response, error.body);
|
||||
}
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
return errorResponse(500, `Speech request failed: ${message}`);
|
||||
}
|
||||
}
|
||||
@@ -85,6 +85,8 @@ function parseGlmEffortTier(model: string): GlmEffortTier | null {
|
||||
return { baseModel: "glm-5.3", effort: "high", transport: "openai" };
|
||||
case "glm-5.3-low":
|
||||
return { baseModel: "glm-5.3", effort: "low", transport: "openai" };
|
||||
case "glm-5.3-max":
|
||||
return { baseModel: "glm-5.3", effort: "max", transport: "openai" };
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import {
|
||||
BaseExecutor,
|
||||
type ExecuteInput,
|
||||
@@ -10,7 +11,7 @@ import {
|
||||
injectReasoningContentForThinkingModel,
|
||||
isThinkingMessageModel,
|
||||
} from "../utils/reasoningContentInjector.ts";
|
||||
import { runWithProxyContext } from "../utils/proxyFetch.ts";
|
||||
import { runWithDirectFetchContext, runWithProxyContext } from "../utils/proxyFetch.ts";
|
||||
import { forwardOpencodeClientHeaders } from "../utils/opencodeHeaders.ts";
|
||||
import {
|
||||
type AccountProxyConfig,
|
||||
@@ -245,6 +246,17 @@ export function createMuseSparkStreamFinishNormalizer(
|
||||
};
|
||||
}
|
||||
|
||||
function isResponsesTerminalLine(line: string): boolean {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed.startsWith("data:")) return false;
|
||||
try {
|
||||
const payload = JSON.parse(trimmed.slice(5).trim()) as Record<string, unknown>;
|
||||
return payload.type === "response.completed";
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
export class OpencodeExecutor extends BaseExecutor {
|
||||
/** Delegates to `isPremiumOpencodeModel`. Exported for testability. */
|
||||
static isPremiumModel(model: string, provider: string): boolean {
|
||||
@@ -384,24 +396,51 @@ export class OpencodeExecutor extends BaseExecutor {
|
||||
const encoder = new TextEncoder();
|
||||
let buffer = "";
|
||||
const reader = response.body.getReader();
|
||||
let closed = false;
|
||||
const stream = new ReadableStream<Uint8Array>({
|
||||
async pull(controller) {
|
||||
async start(controller) {
|
||||
try {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) {
|
||||
if (buffer.length > 0) controller.enqueue(encoder.encode(normalizer(buffer)));
|
||||
controller.close();
|
||||
return;
|
||||
while (!closed) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) {
|
||||
buffer += decoder.decode();
|
||||
if (buffer.length > 0 && !closed) {
|
||||
controller.enqueue(encoder.encode(normalizer(buffer)));
|
||||
}
|
||||
if (!closed) {
|
||||
closed = true;
|
||||
controller.close();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() ?? "";
|
||||
for (const line of lines) {
|
||||
const normalized = normalizer(line);
|
||||
controller.enqueue(encoder.encode(normalized + "\n"));
|
||||
if (isResponsesTerminalLine(line)) {
|
||||
// OpenCode Zen sends a ping after response.completed and may keep
|
||||
// the HTTP connection alive. The Responses terminal event is
|
||||
// authoritative; do not let those post-completion pings hold Chat
|
||||
// Completions open.
|
||||
closed = true;
|
||||
void reader.cancel().catch(() => undefined);
|
||||
controller.close();
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
const lines = buffer.split("\n");
|
||||
buffer = lines.pop() ?? "";
|
||||
for (const line of lines) controller.enqueue(encoder.encode(normalizer(line) + "\n"));
|
||||
} catch (err) {
|
||||
controller.error(err);
|
||||
if (!closed) {
|
||||
closed = true;
|
||||
controller.error(err);
|
||||
}
|
||||
}
|
||||
},
|
||||
cancel(reason) {
|
||||
closed = true;
|
||||
reader.cancel(reason).catch(() => undefined);
|
||||
},
|
||||
});
|
||||
@@ -450,7 +489,10 @@ export class OpencodeExecutor extends BaseExecutor {
|
||||
// 200s ("Provider returned empty content"). Raise tiny budgets to the
|
||||
// floor before dispatch (see MUSE_SPARK_MIN_OUTPUT_TOKENS).
|
||||
if (input.body && typeof input.body === "object" && !Array.isArray(input.body)) {
|
||||
applyMuseSparkMinOutputTokens(String(input.model ?? ""), input.body as Record<string, unknown>);
|
||||
applyMuseSparkMinOutputTokens(
|
||||
String(input.model ?? ""),
|
||||
input.body as Record<string, unknown>
|
||||
);
|
||||
}
|
||||
|
||||
this.syncAccountsFromCredentials(input.credentials);
|
||||
@@ -463,7 +505,9 @@ export class OpencodeExecutor extends BaseExecutor {
|
||||
// else passes untouched: this path deliberately preserves BaseExecutor's
|
||||
// intra-URL 429 retries (no skipUpstreamRetry here).
|
||||
if (this.accounts.length === 1 && !hasProxies) {
|
||||
const single = (await super.execute(input)) as HttpExecuteResult;
|
||||
const single = (await runWithDirectFetchContext(() =>
|
||||
super.execute(input)
|
||||
)) as HttpExecuteResult;
|
||||
if (single.response.status === 400) {
|
||||
let bodyText: string | null = null;
|
||||
try {
|
||||
@@ -630,10 +674,7 @@ export class OpencodeExecutor extends BaseExecutor {
|
||||
}
|
||||
|
||||
// All accounts returned 429 (or errored) — surface the last response.
|
||||
return this.normalizeMuseSparkResponse(
|
||||
input,
|
||||
lastResult ?? (await super.execute(input))
|
||||
);
|
||||
return this.normalizeMuseSparkResponse(input, lastResult ?? (await super.execute(input)));
|
||||
} finally {
|
||||
this._requestFormat = null;
|
||||
}
|
||||
@@ -735,6 +776,18 @@ export class OpencodeExecutor extends BaseExecutor {
|
||||
});
|
||||
}
|
||||
|
||||
// Muse's Responses endpoint rejects the short conversation fingerprint used
|
||||
// by the Chat endpoint in practice. Keep the workaround scoped to Muse.
|
||||
if (
|
||||
this._requestFormat === "openai-responses" &&
|
||||
model.startsWith("muse-spark") &&
|
||||
!/^[0-9a-f]{8}-[0-9a-f]{4}-[1-5][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i.test(
|
||||
headers["x-opencode-session"] || ""
|
||||
)
|
||||
) {
|
||||
headers["x-opencode-session"] = randomUUID();
|
||||
}
|
||||
|
||||
void model;
|
||||
|
||||
return headers;
|
||||
|
||||
@@ -156,13 +156,13 @@ export function pcmToWav(
|
||||
return Buffer.concat([header, pcm]);
|
||||
}
|
||||
|
||||
function parseSampleRate(mimeType: string | undefined): number {
|
||||
export function parsePcmSampleRate(mimeType: string | undefined): number {
|
||||
if (!mimeType) return 24000;
|
||||
const match = /rate=(\d+)/i.exec(mimeType);
|
||||
return match ? parseInt(match[1], 10) : 24000;
|
||||
}
|
||||
|
||||
function extractInlineAudio(
|
||||
export function extractInlineAudio(
|
||||
data: unknown
|
||||
): { base64: string; mimeType: string } | null {
|
||||
const parts = (data as { candidates?: Array<{ content?: { parts?: unknown[] } }> })?.candidates?.[0]
|
||||
@@ -215,7 +215,7 @@ export async function vertexGenerateSpeech(
|
||||
const inline = extractInlineAudio(data);
|
||||
if (!inline) throw new Error("Vertex TTS returned no audio content");
|
||||
const pcm = Buffer.from(inline.base64, "base64");
|
||||
return { audio: pcmToWav(pcm, parseSampleRate(inline.mimeType)), contentType: "audio/wav" };
|
||||
return { audio: pcmToWav(pcm, parsePcmSampleRate(inline.mimeType)), contentType: "audio/wav" };
|
||||
}
|
||||
|
||||
/** Gemini transcription (audio → text). `audioBase64` is the raw file bytes, base64-encoded. */
|
||||
|
||||
@@ -21,6 +21,7 @@ import { getSpeechProvider, parseSpeechModel } from "../config/audioRegistry.ts"
|
||||
import { buildAuthHeaders } from "../config/registryUtils.ts";
|
||||
import { kieExecutor } from "../executors/kie.ts";
|
||||
import { vertexGenerateSpeech } from "../executors/vertexMedia.ts";
|
||||
import { handleGeminiTtsSpeech } from "../executors/geminiTts.ts";
|
||||
import { handleAwsPollySpeech } from "../executors/awsPollyTts.ts";
|
||||
import { handleEdgeTtsSpeech } from "../executors/edgeTts.ts";
|
||||
import { GttsUpstreamError, normalizeGttsLang, synthesizeGtts } from "../executors/gtts.ts";
|
||||
@@ -889,6 +890,13 @@ export async function handleAudioSpeech({
|
||||
headers: { ...CORS_HEADERS, "Content-Type": contentType },
|
||||
});
|
||||
}
|
||||
if (providerConfig.format === "gemini-tts") {
|
||||
return handleGeminiTtsSpeech(credentials, {
|
||||
model: modelId,
|
||||
text: body.input,
|
||||
voice: body.voice,
|
||||
});
|
||||
}
|
||||
|
||||
if (providerConfig.format === "hyperbolic") {
|
||||
return handleHyperbolicSpeech(providerConfig, body, token);
|
||||
|
||||
@@ -91,8 +91,20 @@ interface KieImageOptions {
|
||||
} | null;
|
||||
}
|
||||
|
||||
// KIE Market catalog ids are namespaced for OmniRoute's catalog
|
||||
// (`google-imagen/<model>`), but the KIE Market createTask API expects
|
||||
// vendor-specific upstream ids that do not follow a single consistent
|
||||
// pattern (confirmed against docs.kie.ai/market/google/* — see #11225,
|
||||
// #11296): nano-banana-2 and nano-banana-pro drop the vendor namespace
|
||||
// entirely, while nano-banana and nano-banana-edit use a `google/` prefix
|
||||
// instead of `google-imagen/`. Every other KIE Market namespace (seedream,
|
||||
// flux, ideogram, qwen, wan, grok-imagine, gpt) already matches its real
|
||||
// upstream id byte-for-byte, so this map stays scoped to google-imagen.
|
||||
export const KIE_MARKET_UPSTREAM_MODEL_IDS: ReadonlyMap<string, string> = new Map([
|
||||
["google-imagen/nano-banana", "google/nano-banana"],
|
||||
["google-imagen/nano-banana-2", "nano-banana-2"],
|
||||
["google-imagen/nano-banana-pro", "nano-banana-pro"],
|
||||
["google-imagen/nano-banana-edit", "google/nano-banana-edit"],
|
||||
]);
|
||||
|
||||
export function resolveKieMarketUpstreamModelId(publicModelId: string): string {
|
||||
|
||||
@@ -1008,27 +1008,63 @@ async function captureViaCdp(opts: {
|
||||
}
|
||||
}
|
||||
|
||||
function killProcessTree(child: ChildProcess | null): void {
|
||||
/**
|
||||
* Terminate a spawned browser process and all of its descendants.
|
||||
*
|
||||
* Windows uses `taskkill /pid <pid> /T /F` to walk the process tree and terminate descendants.
|
||||
* Linux/POSIX sends SIGTERM/SIGKILL to the process group (`-pid`) when detached/group leader,
|
||||
* falling back to direct child kill if the process group is unavailable.
|
||||
*/
|
||||
export function killProcessTree(
|
||||
child:
|
||||
| ChildProcess
|
||||
| { pid?: number; kill?: (signal?: NodeJS.Signals | number | string) => boolean | void }
|
||||
| null
|
||||
| undefined,
|
||||
options?: {
|
||||
platform?: string;
|
||||
processKill?: (pid: number, signal?: NodeJS.Signals | string) => void;
|
||||
spawnFn?: typeof spawn;
|
||||
}
|
||||
): void {
|
||||
if (!child?.pid) return;
|
||||
const pid = child.pid;
|
||||
// Never taskkill our own Node/pkg process or its parent (would kill the backend mid-login).
|
||||
if (pid === process.pid || (typeof process.ppid === "number" && pid === process.ppid)) {
|
||||
return;
|
||||
}
|
||||
const platform = options?.platform || process.platform;
|
||||
const processKill = options?.processKill || process.kill.bind(process);
|
||||
const spawnFn = options?.spawnFn || spawn;
|
||||
|
||||
try {
|
||||
if (process.platform === "win32") {
|
||||
if (platform === "win32") {
|
||||
// /T kills only this PID's descendants — not system Chrome profiles we did not spawn.
|
||||
const killer = spawn("taskkill", ["/pid", String(pid), "/T", "/F"], {
|
||||
const killer = spawnFn("taskkill", ["/pid", String(pid), "/T", "/F"], {
|
||||
stdio: "ignore",
|
||||
windowsHide: true,
|
||||
detached: true,
|
||||
});
|
||||
killer.unref?.();
|
||||
killer?.unref?.();
|
||||
} else {
|
||||
child.kill("SIGTERM");
|
||||
let killedGroup = false;
|
||||
try {
|
||||
processKill(-pid, "SIGTERM");
|
||||
killedGroup = true;
|
||||
} catch {
|
||||
try {
|
||||
child.kill?.("SIGTERM");
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
setTimeout(() => {
|
||||
try {
|
||||
child.kill("SIGKILL");
|
||||
if (killedGroup) {
|
||||
processKill(-pid, "SIGKILL");
|
||||
} else {
|
||||
child.kill?.("SIGKILL");
|
||||
}
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
@@ -1036,7 +1072,7 @@ function killProcessTree(child: ChildProcess | null): void {
|
||||
}
|
||||
} catch {
|
||||
try {
|
||||
child.kill();
|
||||
child.kill?.();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
@@ -1175,12 +1211,15 @@ async function runAdobeFireflyCdpBrowser(opts: {
|
||||
// detach so a long Forter wait does not pin the Node process refcount.
|
||||
// Host job SILENT_BREAKAWAY_OK still prevents Chrome from joining the backend job
|
||||
// (that was killing/wedging VibeProxyServices on Sign in with browser).
|
||||
// On POSIX: detached creates a new process group leader so killProcessTree(-pid)
|
||||
// can terminate Chrome and all its child processes (zygote/renderer/GPU).
|
||||
const isDetached = process.platform !== "win32" || !opts.interactive;
|
||||
child = spawn(browserPath, args, {
|
||||
stdio: "ignore",
|
||||
// Interactive sign-in: show Chrome. Background warm: hide spawn console/window
|
||||
// host; headless flags already suppress the browser UI.
|
||||
windowsHide: !opts.interactive,
|
||||
detached: !opts.interactive,
|
||||
detached: isDetached,
|
||||
});
|
||||
if (!opts.interactive) {
|
||||
try {
|
||||
|
||||
@@ -52,8 +52,23 @@ export function preferAntigravityConnectionsWithStoredProject<T extends Record<s
|
||||
const projectId = (psd as Record<string, unknown>).projectId;
|
||||
return typeof projectId === "string" && projectId.trim().length > 0;
|
||||
};
|
||||
const withStoredProject = connections.filter(hasStoredProject);
|
||||
return withStoredProject.length > 0 ? withStoredProject : connections;
|
||||
// #11284: rows whose missing Cloud Code project was CONFIRMED at request
|
||||
// time (errorCode="missing_project_id") are dead weight — drop them when a
|
||||
// healthier sibling exists. When every row is confirmed missing, keep the
|
||||
// pool so the typed 422 (not an empty-selection 404) explains what to fix.
|
||||
const hasHealthySibling = (connection: T): boolean =>
|
||||
connections.some(
|
||||
(other) => other !== connection && other.errorCode !== "missing_project_id"
|
||||
);
|
||||
const candidates = connections.filter(
|
||||
(connection) =>
|
||||
connection.errorCode !== "missing_project_id" ||
|
||||
!hasHealthySibling(connection) ||
|
||||
!hasStoredProject(connection)
|
||||
);
|
||||
const withStoredProject = candidates.filter(hasStoredProject);
|
||||
if (withStoredProject.length > 0) return withStoredProject;
|
||||
return candidates.length > 0 ? candidates : connections;
|
||||
}
|
||||
|
||||
export async function persistDiscoveredAntigravityProjectId(
|
||||
|
||||
@@ -64,6 +64,11 @@ export function persistDiscoveredAntigravityProjectId(
|
||||
errorCode: null,
|
||||
lastError: null,
|
||||
lastErrorType: null,
|
||||
// #11284: a discovered project proves the account is usable again —
|
||||
// re-enable it (markAntigravityMissingCloudCodeProject may have disabled
|
||||
// it after a confirmed-missing 422).
|
||||
isActive: true,
|
||||
testStatus: "active",
|
||||
providerSpecificData,
|
||||
})
|
||||
.catch(() => {})
|
||||
@@ -77,7 +82,14 @@ export function markAntigravityMissingCloudCodeProject(
|
||||
): void {
|
||||
if (!connectionId) return;
|
||||
|
||||
// #11284: a CONFIRMED missing Cloud Code project is not transient — disable
|
||||
// the row so selection rotates to healthy siblings instead of re-dispatching
|
||||
// into the same 422 every request. "unavailable" is deliberately NOT a
|
||||
// terminal status: persistDiscoveredAntigravityProjectId() re-enables the
|
||||
// account the moment a project shows up at request time.
|
||||
void updateProviderConnection(connectionId, {
|
||||
isActive: false,
|
||||
testStatus: "unavailable",
|
||||
errorCode: "missing_project_id",
|
||||
lastError:
|
||||
"Missing Google projectId for Antigravity account. Reconnect OAuth after completing Gemini Code Assist onboarding.",
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import type { ModelCapabilityResolutionSnapshot } from "@/lib/modelCapabilities";
|
||||
|
||||
import type { AutoVariant } from "./autoPrefix";
|
||||
import { VALID_VARIANTS } from "./autoPrefix";
|
||||
import type { PreparedVirtualAutoComboInputs } from "./virtualFactory";
|
||||
@@ -119,8 +121,7 @@ export function isPaidTierAutoId(autoId: string): boolean {
|
||||
* a candidate filter so the virtual combo only scores vision-capable models.
|
||||
*/
|
||||
export type BuiltinAutoSpec =
|
||||
| { variant: AutoVariant | undefined }
|
||||
| { category: AutoCategory; tier?: AutoTier };
|
||||
{ variant: AutoVariant | undefined } | { category: AutoCategory; tier?: AutoTier };
|
||||
|
||||
/**
|
||||
* Vision-flavored flat ids that MUST resolve to the `vision` category (candidate
|
||||
@@ -159,9 +160,14 @@ export function resolveBuiltinAutoSpec(modelStr: string, suffix: string): Builti
|
||||
return { variant: undefined };
|
||||
}
|
||||
|
||||
export async function prepareBuiltinAutoComboInputs(): Promise<PreparedVirtualAutoComboInputs> {
|
||||
export async function prepareBuiltinAutoComboInputs(
|
||||
resolutionSnapshot?: ModelCapabilityResolutionSnapshot
|
||||
): Promise<PreparedVirtualAutoComboInputs> {
|
||||
const { prepareVirtualAutoComboInputs } = await import("./virtualFactory.ts");
|
||||
return prepareVirtualAutoComboInputs({ includeResolvedCapabilities: true });
|
||||
return prepareVirtualAutoComboInputs({
|
||||
includeResolvedCapabilities: true,
|
||||
resolutionSnapshot,
|
||||
});
|
||||
}
|
||||
|
||||
export async function createBuiltinAutoCombo(
|
||||
|
||||
@@ -404,7 +404,9 @@ export function computeAdvertisedLimits(candidates: AdvertisedLimitCandidate[]):
|
||||
return { contextLength, maxOutputTokens };
|
||||
}
|
||||
|
||||
const PREPARED_CAPABILITY_YIELD_INTERVAL = 16;
|
||||
// Catalog-scale pools can contain hundreds of models. Keep both candidate construction
|
||||
// and capability preparation cooperative instead of monopolising one event-loop turn.
|
||||
const VIRTUAL_AUTO_PREPARATION_YIELD_INTERVAL = 4;
|
||||
|
||||
type PreparedCapabilityValues = {
|
||||
resolvedContextLength: number | null;
|
||||
@@ -468,7 +470,7 @@ async function attachPreparedCapabilityValues(
|
||||
};
|
||||
byModel.set(candidate.model, values);
|
||||
state.resolvedSinceYield++;
|
||||
if (state.resolvedSinceYield >= PREPARED_CAPABILITY_YIELD_INTERVAL) {
|
||||
if (state.resolvedSinceYield >= VIRTUAL_AUTO_PREPARATION_YIELD_INTERVAL) {
|
||||
state.resolvedSinceYield = 0;
|
||||
await yieldVirtualAutoPreparationTurn();
|
||||
}
|
||||
@@ -479,7 +481,10 @@ async function attachPreparedCapabilityValues(
|
||||
}
|
||||
|
||||
export async function prepareVirtualAutoComboInputs(
|
||||
options: { includeResolvedCapabilities?: boolean } = {}
|
||||
options: {
|
||||
includeResolvedCapabilities?: boolean;
|
||||
resolutionSnapshot?: ModelCapabilityResolutionSnapshot;
|
||||
} = {}
|
||||
): Promise<PreparedVirtualAutoComboInputs> {
|
||||
const [connections, disabledNoAuthConnections, settings] = await Promise.all([
|
||||
getCachedProviderConnections({ isActive: true }) as Promise<VirtualFactoryConn[]>,
|
||||
@@ -524,6 +529,7 @@ export async function prepareVirtualAutoComboInputs(
|
||||
// Build one logical candidate per provider/model and keep account fallback as an
|
||||
// allowlist on that candidate. This avoids both the old "first registry model per
|
||||
// connection" blind spot and a connections × models Cartesian candidate pool.
|
||||
let candidateModelsSinceYield = 0;
|
||||
for (const [providerId, providerConnections] of connectionsByProvider) {
|
||||
const providerInfo = registry[providerId];
|
||||
const registryModelIds = Array.isArray(providerInfo?.models)
|
||||
@@ -557,6 +563,11 @@ export async function prepareVirtualAutoComboInputs(
|
||||
: Array.from(new Set([...registryModelIds, ...defaultModelIds]));
|
||||
|
||||
for (const modelId of modelIds) {
|
||||
candidateModelsSinceYield++;
|
||||
if (candidateModelsSinceYield >= VIRTUAL_AUTO_PREPARATION_YIELD_INTERVAL) {
|
||||
candidateModelsSinceYield = 0;
|
||||
await yieldVirtualAutoPreparationTurn();
|
||||
}
|
||||
if (hiddenModels?.has(modelId)) continue;
|
||||
|
||||
const allowedConnectionIds = providerConnections
|
||||
@@ -655,7 +666,7 @@ export async function prepareVirtualAutoComboInputs(
|
||||
const capabilityState: PreparedCapabilityState = {
|
||||
byTarget: new Map(),
|
||||
resolvedSinceYield: 0,
|
||||
resolutionSnapshot: createModelCapabilityResolutionSnapshot(),
|
||||
resolutionSnapshot: options.resolutionSnapshot ?? createModelCapabilityResolutionSnapshot(),
|
||||
};
|
||||
return {
|
||||
regularCandidates: await attachPreparedCapabilityValues(regularCandidates, capabilityState),
|
||||
|
||||
@@ -210,12 +210,16 @@ import {
|
||||
normalizeConnectionStatus,
|
||||
hasFutureRateLimitUntil,
|
||||
getConnectionStatusQuotaCutoffReason,
|
||||
getPersistedConnectionCooldownSkipReason,
|
||||
resolvePersistedConnectionCooldownSkipReason,
|
||||
isContextOverflow400,
|
||||
isParamValidation400,
|
||||
isModelScoped400,
|
||||
} from "./combo/comboPredicates.ts";
|
||||
export {
|
||||
getConnectionStatusQuotaCutoffReason,
|
||||
getPersistedConnectionCooldownSkipReason,
|
||||
resolvePersistedConnectionCooldownSkipReason,
|
||||
isContextOverflow400,
|
||||
isParamValidation400,
|
||||
isModelScoped400,
|
||||
@@ -320,6 +324,26 @@ export {
|
||||
* peekStickyConnectionId guards against clearing an unrelated pin when the
|
||||
* failing target isn't actually the currently sticky-bound connection.
|
||||
*/
|
||||
/**
|
||||
* Connection read for the pre-dispatch persisted-cooldown gate.
|
||||
*
|
||||
* `fresh: false` (first attempt) uses the shared 5s readCache — the row was just
|
||||
* read by the surrounding target resolution, so a second uncached hit is pure cost.
|
||||
* `fresh: true` (every retry) goes straight to SQLite: during a burst a sibling
|
||||
* request routinely writes `rate_limited_until` while this attempt is sleeping out
|
||||
* its retry delay, so the cached snapshot would still say "no cooldown" — which is
|
||||
* exactly how a retry ended up dispatching into a real upstream 429 on a connection
|
||||
* the engine had already marked unavailable.
|
||||
*/
|
||||
async function readConnectionForCooldownGate(
|
||||
connectionId: string,
|
||||
fresh: boolean
|
||||
): Promise<Record<string, unknown> | null | undefined> {
|
||||
if (!fresh) return getCachedProviderConnectionById(connectionId);
|
||||
const { getProviderConnectionById } = await import("@/lib/db/providers");
|
||||
return (await getProviderConnectionById(connectionId)) as Record<string, unknown> | null;
|
||||
}
|
||||
|
||||
export function releaseStickyPinOnFailure(
|
||||
messageHash: string | null | undefined,
|
||||
failedConnectionId: string | null | undefined
|
||||
@@ -1214,6 +1238,23 @@ async function handleComboChatInner({
|
||||
}
|
||||
: { ...target, modelAbortSignal: abortControllers.get(i)!.signal };
|
||||
|
||||
// Persist the connection cooldown before dispatch. AUTH only learns
|
||||
// unavailable during credential lookup, so a burst would otherwise
|
||||
// burn max_concurrent slots on real upstream calls against a row
|
||||
// SQLite already locked until the reset.
|
||||
if (target.connectionId && !allowRateLimitedConnection) {
|
||||
const persistedSkip = await resolvePersistedConnectionCooldownSkipReason(
|
||||
target,
|
||||
(id) => readConnectionForCooldownGate(id, false),
|
||||
allowRateLimitedConnection
|
||||
);
|
||||
if (persistedSkip) {
|
||||
log.info("COMBO", persistedSkip);
|
||||
if (i > 0) fallbackCount++;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// #1731 / #1731v2: skip targets already known-exhausted this request (shared predicate).
|
||||
const exhaustedSkip = getExhaustedTargetSkipReason(
|
||||
target,
|
||||
@@ -1471,6 +1512,21 @@ async function handleComboChatInner({
|
||||
log.info("COMBO", `Client disconnected during retry delay — aborting`);
|
||||
return { ok: false, response: errorResponse(499, "Client disconnected") };
|
||||
}
|
||||
|
||||
// Retry re-check: a sibling attempt (or attempt 1) may have persisted
|
||||
// a quota cooldown while this attempt was sleeping out its retry delay
|
||||
// ("Trying model 1/7: zai/glm-5.3 (retry 1)" after "already marked
|
||||
// unavailable until …"). Reads fresh, not cached: see readConnectionForCooldownGate.
|
||||
const persistedRetrySkip = await resolvePersistedConnectionCooldownSkipReason(
|
||||
target,
|
||||
(id) => readConnectionForCooldownGate(id, true),
|
||||
allowRateLimitedConnection
|
||||
);
|
||||
if (persistedRetrySkip) {
|
||||
log.info("COMBO", persistedRetrySkip);
|
||||
if (i > 0) fallbackCount++;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
log.info(
|
||||
|
||||
@@ -482,6 +482,73 @@ export function getConnectionStatusQuotaCutoffReason(
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-dispatch skip for a combo target whose connection is already on a
|
||||
* persisted cooldown. Combo previously only learned that from AUTH after a
|
||||
* real upstream call, so a burst could burn max_concurrent slots against a
|
||||
* connection that SQLite already marked unavailable until a future reset.
|
||||
*
|
||||
* Honours a future rateLimitedUntil regardless of testStatus, the terminal
|
||||
* statuses that must never be dispatched, and a bare `unavailable` status even
|
||||
* when no timestamp was written alongside it.
|
||||
*/
|
||||
export function getPersistedConnectionCooldownSkipReason(
|
||||
target: { modelStr: string; connectionId?: string | null },
|
||||
connection: Record<string, unknown> | null | undefined,
|
||||
allowRateLimitedConnection = false
|
||||
): string | null {
|
||||
if (allowRateLimitedConnection) return null;
|
||||
if (!target.connectionId || !connection) return null;
|
||||
if (hasFutureRateLimitUntil(connection.rateLimitedUntil)) {
|
||||
return `Skipping ${target.modelStr} — connection ${target.connectionId} has persisted cooldown until ${String(connection.rateLimitedUntil)}`;
|
||||
}
|
||||
const status = normalizeConnectionStatus(connection.testStatus);
|
||||
if (QUOTA_BLOCKING_CONNECTION_STATUSES.has(status)) {
|
||||
return `Skipping ${target.modelStr} — connection ${target.connectionId} status=${status}`;
|
||||
}
|
||||
// `unavailable` with no (or an already-expired) rateLimitedUntil still means AUTH
|
||||
// took this connection out of rotation — markAccountUnavailable() writes the status
|
||||
// before, and sometimes without, a timestamp ("Using zai account …" then a real
|
||||
// upstream 429). Without this branch the pre-skip only fired once the timestamp had
|
||||
// landed, so a burst still dispatched against a connection AUTH had already retired.
|
||||
// Lazy recovery is unaffected: clearAccountError() resets the status on first success.
|
||||
if (status === "unavailable") {
|
||||
return `Skipping ${target.modelStr} — connection ${target.connectionId} status=unavailable`;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Async wrapper around `getPersistedConnectionCooldownSkipReason` for the combo
|
||||
* dispatchers, which must re-check the persisted cooldown before EVERY upstream
|
||||
* attempt — not just once before the retry loop.
|
||||
*
|
||||
* The retry path is exactly where the stale-read risk lives: a sibling request in
|
||||
* the same burst can write `rate_limited_until` while this attempt is sleeping out
|
||||
* its retry delay, so the caller passes a cache-bypassing fetcher for retry > 0
|
||||
* (the readCache TTL is 5s, long enough to serve a "no cooldown" snapshot written
|
||||
* before the 429 landed).
|
||||
*
|
||||
* Kept dependency-free — the fetcher is injected, so this module stays pure and
|
||||
* unit-testable without a DB.
|
||||
*/
|
||||
export async function resolvePersistedConnectionCooldownSkipReason(
|
||||
target: { modelStr: string; connectionId?: string | null },
|
||||
fetchConnection: (id: string) => Promise<Record<string, unknown> | null | undefined>,
|
||||
allowRateLimitedConnection = false
|
||||
): Promise<string | null> {
|
||||
if (allowRateLimitedConnection) return null;
|
||||
if (!target.connectionId) return null;
|
||||
let connection: Record<string, unknown> | null | undefined;
|
||||
try {
|
||||
connection = await fetchConnection(target.connectionId);
|
||||
} catch {
|
||||
// A DB read failure must never block dispatch — fall through to the upstream call.
|
||||
return null;
|
||||
}
|
||||
return getPersistedConnectionCooldownSkipReason(target, connection, allowRateLimitedConnection);
|
||||
}
|
||||
|
||||
/** @param {string} errorText */
|
||||
export function isContextOverflow400(errorText: string | null | undefined): boolean {
|
||||
const text = String(errorText || "");
|
||||
|
||||
@@ -290,6 +290,7 @@ export function shouldProtectOriginalFirst(
|
||||
return (
|
||||
stickyStuck ||
|
||||
autoUsedExplicitRouter ||
|
||||
strategy === "auto" ||
|
||||
strategy === "quota-share" ||
|
||||
strategy === "weighted" ||
|
||||
strategy === "priority" ||
|
||||
|
||||
@@ -64,6 +64,7 @@ export function compressAggressive(
|
||||
let summarizerSavings = 0;
|
||||
let toolResultSavings = 0;
|
||||
let agingSavings = 0;
|
||||
const lastUserIdx = currentMessages.findLastIndex((m) => m.role === "user");
|
||||
|
||||
// Step 1: Tool-result compression
|
||||
try {
|
||||
@@ -110,7 +111,8 @@ export function compressAggressive(
|
||||
currentMessages,
|
||||
cfg.thresholds,
|
||||
summarizer,
|
||||
cfg.preserveSystemPrompt !== false
|
||||
cfg.preserveSystemPrompt !== false,
|
||||
lastUserIdx
|
||||
);
|
||||
agingSavings = agingResult.saved;
|
||||
currentMessages = agingResult.messages as ChatMessage[];
|
||||
@@ -121,8 +123,9 @@ export function compressAggressive(
|
||||
// Step 3: Fallback summarizer for remaining long messages
|
||||
if (cfg.summarizerEnabled) {
|
||||
try {
|
||||
currentMessages = currentMessages.map((msg) => {
|
||||
currentMessages = currentMessages.map((msg, idx) => {
|
||||
if (cfg.preserveSystemPrompt !== false && msg.role === "system") return msg;
|
||||
if (idx === lastUserIdx) return msg;
|
||||
const text = extractTextContent(msg.content);
|
||||
if (!text || COMPRESSED_MARKER_RE.test(text)) return msg;
|
||||
if (text.length <= cfg.maxTokensPerMessage * 4) return msg;
|
||||
@@ -133,7 +136,10 @@ export function compressAggressive(
|
||||
});
|
||||
if (summary && summary.length < text.length) {
|
||||
summarizerSavings += estimateTokens(text) - estimateTokens(summary);
|
||||
return setContent(msg, `[COMPRESSED:summary] ${summary}`);
|
||||
const finalSummary = COMPRESSED_MARKER_RE.test(summary)
|
||||
? summary
|
||||
: `[COMPRESSED:summary] ${summary}`;
|
||||
return setContent(msg, finalSummary);
|
||||
}
|
||||
return msg;
|
||||
});
|
||||
@@ -153,13 +159,27 @@ export function compressAggressive(
|
||||
|
||||
if (resultStats.savingsPercent < cfg.minSavingsThreshold * 100) {
|
||||
try {
|
||||
const cavemanResult = cavemanCompress({ messages: currentMessages as unknown as Parameters<typeof cavemanCompress>[0]["messages"] });
|
||||
if (cavemanResult?.compressed && cavemanResult.stats) {
|
||||
const cavemanSavings = cavemanResult.stats.savingsPercent ?? 0;
|
||||
if (cavemanSavings > resultStats.savingsPercent) {
|
||||
currentMessages = (cavemanResult.body?.messages ?? currentMessages) as ChatMessage[];
|
||||
resultStats.compressedTokens = cavemanResult.stats.compressedTokens ?? compressedTokens;
|
||||
resultStats.savingsPercent = cavemanSavings;
|
||||
const cavemanResult = cavemanCompress(
|
||||
{
|
||||
messages: currentMessages as unknown as Parameters<typeof cavemanCompress>[0]["messages"],
|
||||
},
|
||||
{ enabled: true }
|
||||
);
|
||||
if (cavemanResult?.compressed && cavemanResult.body?.messages) {
|
||||
const rawMsgs = cavemanResult.body.messages as ChatMessage[];
|
||||
const candidateMsgs = rawMsgs.map((msg, idx) =>
|
||||
idx === lastUserIdx ? currentMessages[idx] : msg
|
||||
);
|
||||
const candidateTokens = candidateMsgs.reduce(
|
||||
(sum, m) => sum + estimateTokens(extractTextContent(m.content)),
|
||||
0
|
||||
);
|
||||
const candidateSavings =
|
||||
originalTokens > 0 ? ((originalTokens - candidateTokens) / originalTokens) * 100 : 0;
|
||||
if (candidateSavings > resultStats.savingsPercent) {
|
||||
currentMessages = candidateMsgs;
|
||||
resultStats.compressedTokens = candidateTokens;
|
||||
resultStats.savingsPercent = candidateSavings;
|
||||
resultStats.techniquesUsed.push("caveman-fallback");
|
||||
}
|
||||
}
|
||||
@@ -172,12 +192,21 @@ export function compressAggressive(
|
||||
{ messages: currentMessages },
|
||||
{ preserveSystemPrompt: cfg.preserveSystemPrompt !== false }
|
||||
);
|
||||
if (liteResult?.compressed && liteResult.stats) {
|
||||
const liteSavings = liteResult.stats.savingsPercent ?? 0;
|
||||
if (liteSavings > resultStats.savingsPercent) {
|
||||
currentMessages = (liteResult.body?.messages ?? currentMessages) as ChatMessage[];
|
||||
resultStats.compressedTokens = liteResult.stats.compressedTokens ?? compressedTokens;
|
||||
resultStats.savingsPercent = liteSavings;
|
||||
if (liteResult?.compressed && liteResult.body?.messages) {
|
||||
const rawMsgs = liteResult.body.messages as ChatMessage[];
|
||||
const candidateMsgs = rawMsgs.map((msg, idx) =>
|
||||
idx === lastUserIdx ? currentMessages[idx] : msg
|
||||
);
|
||||
const candidateTokens = candidateMsgs.reduce(
|
||||
(sum, m) => sum + estimateTokens(extractTextContent(m.content)),
|
||||
0
|
||||
);
|
||||
const candidateSavings =
|
||||
originalTokens > 0 ? ((originalTokens - candidateTokens) / originalTokens) * 100 : 0;
|
||||
if (candidateSavings > resultStats.savingsPercent) {
|
||||
currentMessages = candidateMsgs;
|
||||
resultStats.compressedTokens = candidateTokens;
|
||||
resultStats.savingsPercent = candidateSavings;
|
||||
resultStats.techniquesUsed.push("lite-fallback");
|
||||
}
|
||||
}
|
||||
|
||||
40
open-sse/services/compression/compressionWorker.ts
Normal file
@@ -0,0 +1,40 @@
|
||||
import { parentPort } from "node:worker_threads";
|
||||
import {
|
||||
applyCompression,
|
||||
applyStackedCompression,
|
||||
type StackedCompressionStep,
|
||||
} from "./strategySelector.ts";
|
||||
import type {
|
||||
CompressionWorkerJob,
|
||||
CompressionWorkerMessage,
|
||||
} from "./compressionWorkerProtocol.ts";
|
||||
|
||||
if (!parentPort) throw new Error("compressionWorker must run in a worker thread");
|
||||
parentPort.on("message", (job: CompressionWorkerJob) => {
|
||||
try {
|
||||
const onEngineStep = (step: StackedCompressionStep) =>
|
||||
parentPort.postMessage({
|
||||
id: job.id,
|
||||
type: "step",
|
||||
step,
|
||||
} satisfies CompressionWorkerMessage);
|
||||
const result =
|
||||
job.mode === "stacked"
|
||||
? applyStackedCompression(job.body, job.options?.config?.stackedPipeline, {
|
||||
...job.options,
|
||||
onEngineStep,
|
||||
})
|
||||
: applyCompression(job.body, job.mode, job.options);
|
||||
parentPort.postMessage({
|
||||
id: job.id,
|
||||
type: "result",
|
||||
result,
|
||||
} satisfies CompressionWorkerMessage);
|
||||
} catch (error) {
|
||||
parentPort.postMessage({
|
||||
id: job.id,
|
||||
type: "error",
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
} satisfies CompressionWorkerMessage);
|
||||
}
|
||||
});
|
||||
165
open-sse/services/compression/compressionWorkerPool.ts
Normal file
@@ -0,0 +1,165 @@
|
||||
import { existsSync } from "node:fs";
|
||||
import { dirname, join } from "node:path";
|
||||
import { fileURLToPath, pathToFileURL } from "node:url";
|
||||
import { Worker } from "node:worker_threads";
|
||||
import type { CompressionResult } from "./types.ts";
|
||||
import type { StackedCompressionStep } from "./strategySelector.ts";
|
||||
import type {
|
||||
CompressionWorkerJob,
|
||||
CompressionWorkerMessage,
|
||||
CompressionWorkerOptions,
|
||||
} from "./compressionWorkerProtocol.ts";
|
||||
|
||||
function positiveInteger(value: string | undefined, fallback: number): number {
|
||||
const parsed = Number(value);
|
||||
return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : fallback;
|
||||
}
|
||||
function workerUrl(): URL {
|
||||
const dir = dirname(fileURLToPath(import.meta.url));
|
||||
for (const name of ["compressionWorker.js", "compressionWorker.ts"]) {
|
||||
const candidate = join(dir, name);
|
||||
if (existsSync(candidate)) return pathToFileURL(candidate);
|
||||
}
|
||||
return pathToFileURL(join(dir, "compressionWorker.js"));
|
||||
}
|
||||
function unchanged(body: Record<string, unknown>): CompressionResult {
|
||||
return { body, compressed: false, stats: null };
|
||||
}
|
||||
interface PendingJob extends CompressionWorkerJob {
|
||||
originalBody: Record<string, unknown>;
|
||||
resolve: (result: CompressionResult) => void;
|
||||
onEngineStep?: (step: StackedCompressionStep) => void;
|
||||
}
|
||||
interface PoolWorker {
|
||||
worker: Worker;
|
||||
job: PendingJob | null;
|
||||
timeout: NodeJS.Timeout | null;
|
||||
idle: NodeJS.Timeout | null;
|
||||
}
|
||||
|
||||
export class CompressionWorkerPool {
|
||||
private readonly queue: PendingJob[] = [];
|
||||
private readonly workers = new Set<PoolWorker>();
|
||||
private nextId = 1;
|
||||
private readonly size: number;
|
||||
private readonly timeoutMs: number;
|
||||
private readonly idleMs: number;
|
||||
|
||||
constructor({
|
||||
size = positiveInteger(process.env.OMNI_COMPRESSION_WORKERS, 2),
|
||||
timeoutMs = positiveInteger(process.env.OMNI_COMPRESSION_WORKER_TIMEOUT_MS, 120_000),
|
||||
idleMs = positiveInteger(process.env.OMNI_COMPRESSION_WORKER_IDLE_MS, 60_000),
|
||||
}: { size?: number; timeoutMs?: number; idleMs?: number } = {}) {
|
||||
this.size = Math.max(1, Math.floor(size));
|
||||
this.timeoutMs = Math.max(1, Math.floor(timeoutMs));
|
||||
this.idleMs = Math.max(1, Math.floor(idleMs));
|
||||
}
|
||||
|
||||
run(
|
||||
body: Record<string, unknown>,
|
||||
mode: CompressionWorkerJob["mode"],
|
||||
options?: CompressionWorkerOptions,
|
||||
onEngineStep?: (step: StackedCompressionStep) => void
|
||||
): Promise<CompressionResult> {
|
||||
return new Promise((resolve) => {
|
||||
this.queue.push({
|
||||
id: this.nextId++,
|
||||
body,
|
||||
mode,
|
||||
options,
|
||||
originalBody: body,
|
||||
resolve,
|
||||
onEngineStep,
|
||||
});
|
||||
this.dispatch();
|
||||
});
|
||||
}
|
||||
async close(): Promise<void> {
|
||||
for (const job of this.queue.splice(0)) job.resolve(unchanged(job.originalBody));
|
||||
await Promise.all([...this.workers].map((slot) => this.remove(slot, true)));
|
||||
}
|
||||
private spawn(): PoolWorker {
|
||||
const slot: PoolWorker = {
|
||||
worker: new Worker(workerUrl()),
|
||||
job: null,
|
||||
timeout: null,
|
||||
idle: null,
|
||||
};
|
||||
this.workers.add(slot);
|
||||
slot.worker.on("message", (message: CompressionWorkerMessage) =>
|
||||
this.handleMessage(slot, message)
|
||||
);
|
||||
slot.worker.on("error", () => this.fail(slot));
|
||||
slot.worker.on("exit", () => {
|
||||
if (this.workers.has(slot)) this.fail(slot);
|
||||
});
|
||||
return slot;
|
||||
}
|
||||
private dispatch(): void {
|
||||
while (this.queue.length) {
|
||||
let slot = [...this.workers].find((candidate) => !candidate.job);
|
||||
if (!slot && this.workers.size < this.size) slot = this.spawn();
|
||||
if (!slot) return;
|
||||
if (slot.idle) clearTimeout(slot.idle);
|
||||
const job = this.queue.shift();
|
||||
if (!job) return;
|
||||
slot.job = job;
|
||||
slot.timeout = setTimeout(() => this.fail(slot!), this.timeoutMs);
|
||||
slot.timeout.unref();
|
||||
const { originalBody: _body, resolve: _resolve, onEngineStep: _step, ...wireJob } = job;
|
||||
slot.worker.postMessage(wireJob);
|
||||
}
|
||||
}
|
||||
private handleMessage(slot: PoolWorker, message: CompressionWorkerMessage): void {
|
||||
const job = slot.job;
|
||||
if (!job || job.id !== message.id) return;
|
||||
if (message.type === "step") {
|
||||
try {
|
||||
job.onEngineStep?.(message.step);
|
||||
} catch {
|
||||
// Telemetry is best-effort.
|
||||
}
|
||||
return;
|
||||
}
|
||||
this.finish(slot, message.type === "result" ? message.result : unchanged(job.originalBody));
|
||||
}
|
||||
private finish(slot: PoolWorker, result: CompressionResult): void {
|
||||
const job = slot.job;
|
||||
if (!job) return;
|
||||
if (slot.timeout) clearTimeout(slot.timeout);
|
||||
slot.timeout = null;
|
||||
slot.job = null;
|
||||
job.resolve(result);
|
||||
slot.idle = setTimeout(() => void this.remove(slot, false), this.idleMs);
|
||||
slot.idle.unref();
|
||||
this.dispatch();
|
||||
}
|
||||
private fail(slot: PoolWorker): void {
|
||||
const job = slot.job;
|
||||
if (job) job.resolve(unchanged(job.originalBody));
|
||||
slot.job = null;
|
||||
void this.remove(slot, true).finally(() => this.dispatch());
|
||||
}
|
||||
private async remove(slot: PoolWorker, terminate: boolean): Promise<void> {
|
||||
if (!this.workers.delete(slot)) return;
|
||||
if (slot.timeout) clearTimeout(slot.timeout);
|
||||
if (slot.idle) clearTimeout(slot.idle);
|
||||
if (terminate) await slot.worker.terminate().catch(() => undefined);
|
||||
}
|
||||
}
|
||||
|
||||
let pool: CompressionWorkerPool | null = null;
|
||||
export function runCompressionInWorker(
|
||||
body: Record<string, unknown>,
|
||||
mode: CompressionWorkerJob["mode"],
|
||||
options?: CompressionWorkerOptions,
|
||||
onEngineStep?: (step: StackedCompressionStep) => void
|
||||
): Promise<CompressionResult> {
|
||||
pool ??= new CompressionWorkerPool();
|
||||
return pool.run(body, mode, options, onEngineStep);
|
||||
}
|
||||
export async function closeCompressionWorkerPoolForTests(): Promise<void> {
|
||||
const active = pool;
|
||||
pool = null;
|
||||
await active?.close();
|
||||
}
|
||||
71
open-sse/services/compression/compressionWorkerProtocol.ts
Normal file
@@ -0,0 +1,71 @@
|
||||
import type { CompressionConfig, CompressionMode, CompressionResult } from "./types.ts";
|
||||
import type { StackedCompressionStep } from "./strategySelector.ts";
|
||||
import type {
|
||||
CompressionStage,
|
||||
CompressionWireFormat,
|
||||
ImageTransportFidelity,
|
||||
} from "./engines/types.ts";
|
||||
|
||||
export interface CompressionWorkerOptions {
|
||||
model?: string;
|
||||
supportsVision?: boolean | null;
|
||||
providerTransport?: "direct" | "aggregator";
|
||||
provider?: string;
|
||||
imageTransportFidelity?: ImageTransportFidelity;
|
||||
sourceFormat?: CompressionWireFormat;
|
||||
targetFormat?: CompressionWireFormat;
|
||||
compressionStage?: CompressionStage;
|
||||
config?: CompressionConfig;
|
||||
}
|
||||
export interface CompressionWorkerJob {
|
||||
id: number;
|
||||
body: Record<string, unknown>;
|
||||
mode: CompressionMode;
|
||||
options?: CompressionWorkerOptions;
|
||||
}
|
||||
export type CompressionWorkerMessage =
|
||||
| { id: number; type: "step"; step: StackedCompressionStep }
|
||||
| { id: number; type: "result"; result: CompressionResult }
|
||||
| { id: number; type: "error"; error: string };
|
||||
|
||||
function isPlainObject(value: object): value is Record<string, unknown> {
|
||||
const prototype = Object.getPrototypeOf(value);
|
||||
return prototype === Object.prototype || prototype === null;
|
||||
}
|
||||
export function isStrictlySerializable(value: unknown, seen = new Set<object>()): boolean {
|
||||
if (
|
||||
value === null ||
|
||||
typeof value === "string" ||
|
||||
typeof value === "boolean" ||
|
||||
typeof value === "number"
|
||||
) {
|
||||
return typeof value !== "number" || Number.isFinite(value);
|
||||
}
|
||||
if (typeof value !== "object" || seen.has(value)) return false;
|
||||
seen.add(value);
|
||||
if (Array.isArray(value)) return value.every((entry) => isStrictlySerializable(entry, seen));
|
||||
if (!isPlainObject(value)) return false;
|
||||
return Object.values(value).every((entry) => isStrictlySerializable(entry, seen));
|
||||
}
|
||||
|
||||
const WORKER_STACK_ENGINES = new Set(["caveman", "rtk", "standard"]);
|
||||
export function isCompressionWorkerEligible(
|
||||
body: Record<string, unknown>,
|
||||
mode: CompressionMode,
|
||||
options?: CompressionWorkerOptions
|
||||
): boolean {
|
||||
if (mode !== "standard" && mode !== "rtk" && mode !== "stacked") return false;
|
||||
if (mode === "stacked") {
|
||||
const pipeline = options?.config?.stackedPipeline;
|
||||
if (!Array.isArray(pipeline) || pipeline.length === 0) return false;
|
||||
if (
|
||||
pipeline.some((step) => {
|
||||
const engine = typeof step === "string" ? step : step.engine;
|
||||
return !WORKER_STACK_ENGINES.has(engine);
|
||||
})
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return isStrictlySerializable({ body, mode, ...(options ? { options } : {}) });
|
||||
}
|
||||
@@ -67,7 +67,8 @@ export function applyAging(
|
||||
messages: unknown[],
|
||||
thresholds?: AgingThresholds,
|
||||
summarizer?: Summarizer,
|
||||
preserveSystemPrompt = true
|
||||
preserveSystemPrompt = true,
|
||||
spareUserIndex?: number
|
||||
): { messages: unknown[]; saved: number } {
|
||||
const t = thresholds ?? DEFAULT_AGGRESSIVE_CONFIG.thresholds;
|
||||
const sum = summarizer ?? {
|
||||
@@ -81,6 +82,9 @@ export function applyAging(
|
||||
const typed = messages as ChatMessage[];
|
||||
if (typed.length === 0) return { messages: [], saved: 0 };
|
||||
|
||||
const lastUserIdx =
|
||||
spareUserIndex !== undefined ? spareUserIndex : typed.findLastIndex((m) => m.role === "user");
|
||||
|
||||
const totalMessages = typed.length;
|
||||
const result: ChatMessage[] = [];
|
||||
let saved = 0;
|
||||
@@ -89,7 +93,11 @@ export function applyAging(
|
||||
const msg = typed[i];
|
||||
const text = extractTextContent(msg.content);
|
||||
|
||||
if ((preserveSystemPrompt && msg.role === "system") || COMPRESSED_MARKER_RE.test(text)) {
|
||||
if (
|
||||
(preserveSystemPrompt && msg.role === "system") ||
|
||||
COMPRESSED_MARKER_RE.test(text) ||
|
||||
i === lastUserIdx
|
||||
) {
|
||||
result.push(msg);
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -519,6 +519,28 @@ async function runCompressionAsync(
|
||||
cachingContext?: CachingDetectionContext;
|
||||
}
|
||||
): Promise<CompressionResult> {
|
||||
const workerOptions = options
|
||||
? {
|
||||
model: options.model,
|
||||
supportsVision: options.supportsVision,
|
||||
providerTransport: options.providerTransport,
|
||||
provider: options.provider,
|
||||
imageTransportFidelity: options.imageTransportFidelity,
|
||||
sourceFormat: options.sourceFormat,
|
||||
targetFormat: options.targetFormat,
|
||||
compressionStage: options.compressionStage,
|
||||
config: options.config,
|
||||
}
|
||||
: undefined;
|
||||
const { isCompressionWorkerEligible } = await import("./compressionWorkerProtocol.ts");
|
||||
if (isCompressionWorkerEligible(body, mode, workerOptions)) {
|
||||
try {
|
||||
const { runCompressionInWorker } = await import("./compressionWorkerPool.ts");
|
||||
return await runCompressionInWorker(body, mode, workerOptions, options?.onEngineStep);
|
||||
} catch {
|
||||
return { body, compressed: false, stats: null };
|
||||
}
|
||||
}
|
||||
if (
|
||||
options?.config?.memoizeCompressionResults === true &&
|
||||
// Only memoize for an explicit principal — a missing principalId would collapse
|
||||
|
||||
@@ -18,6 +18,7 @@ import {
|
||||
TokenExtractionConfig,
|
||||
type TokenSource,
|
||||
} from "./tokenExtractionConfig";
|
||||
import { matchesCookieDomain } from "../utils/cookieDomain";
|
||||
|
||||
// ─── Types ──────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -196,9 +197,14 @@ export class InAppLoginService extends EventEmitter {
|
||||
for (const source of tokenSources) {
|
||||
if (source.type === "cookie") {
|
||||
const domain = source.domain || undefined;
|
||||
// Exact host or dot-boundary suffix, never `includes()`: a cookie
|
||||
// from `<domain>.attacker.tld` would otherwise be captured and
|
||||
// persisted as the operator's credential. Same class CodeQL flagged
|
||||
// in volcengineConsoleAutoLogin (#860/#861); this callsite was not
|
||||
// flagged because the expected domain is config-supplied.
|
||||
const matched = cookies.find(
|
||||
(c: any) =>
|
||||
c.name === source.name && (!domain || c.domain.includes(domain.replace(/^\./, "")))
|
||||
c.name === source.name && (!domain || matchesCookieDomain(c.domain, domain))
|
||||
);
|
||||
if (matched && !credentials[source.name]) {
|
||||
credentials[source.name] = matched.value;
|
||||
|
||||
@@ -26,19 +26,121 @@ export function shouldPreserveQuotaSignals(
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a day-granularity quota reset countdown ("Your quota will reset in
|
||||
* 3 days.", "Resets in 13 days") out of an upstream 429 body.
|
||||
* Parse a day-granularity quota reset countdown (\"Your quota will reset in
|
||||
* 3 days.\", \"Resets in 13 days\") out of an upstream 429 body.
|
||||
*
|
||||
* Companion to the Xh/Ym/Zs countdown parsing already handled inline by
|
||||
* `parseRetryFromErrorText` — none of those patterns match when the upstream
|
||||
* expresses the reset window in whole days rather than hours/minutes/seconds,
|
||||
* so a multi-day quota reset previously parsed to `null` and fell back to the
|
||||
* engine's ~seconds-scale default cooldown.
|
||||
*
|
||||
* Delegates to `parseIsoDateTimeResetMs` (absolute \"reset at YYYY-MM-DD HH:MM:SS\")
|
||||
* and then `parseMonthDayResetMs` (year-less \"reset at MM-DD HH:MM:SS UTC\") so
|
||||
* every absolute-reset shape an upstream uses resolves to the real wait.
|
||||
*/
|
||||
export function parseDayGranularityResetMs(msg: string, maxMs: number): number | null {
|
||||
export function parseDayGranularityResetMs(
|
||||
msg: string,
|
||||
maxMs: number,
|
||||
nowMs: number = Date.now()
|
||||
): number | null {
|
||||
const dayMatch = /reset(?:s)?\s+in\s+(\d+)\s*day(?:s)?/i.exec(msg);
|
||||
if (!dayMatch) return null;
|
||||
const days = Number.parseInt(dayMatch[1], 10);
|
||||
if (!Number.isFinite(days) || days <= 0) return null;
|
||||
return Math.min(days * 24 * 3600 * 1000, maxMs);
|
||||
if (dayMatch) {
|
||||
const days = Number.parseInt(dayMatch[1], 10);
|
||||
if (Number.isFinite(days) && days > 0) {
|
||||
return Math.min(days * 24 * 3600 * 1000, maxMs);
|
||||
}
|
||||
}
|
||||
const isoMs = parseIsoDateTimeResetMs(msg, maxMs, nowMs);
|
||||
if (isoMs !== null) return isoMs;
|
||||
return parseMonthDayResetMs(msg, maxMs, nowMs);
|
||||
}
|
||||
|
||||
/**
|
||||
* Z.AI (GLM) reports an exhausted weekly/monthly cap with a FULL absolute
|
||||
* datetime rather than a countdown:
|
||||
*
|
||||
* \"[1310][Weekly/Monthly Limit Exhausted. … Your limit will reset at
|
||||
* 2026-08-29 21:01:21]\"
|
||||
*
|
||||
* `parseRetryFromErrorText` (accountFallback.ts) has an equivalent ISO matcher,
|
||||
* but `buildWeeklyQuotaFallback` never reaches it: it calls
|
||||
* `parseDayGranularityResetMs` directly, and neither the \"reset in N days\" nor
|
||||
* the year-less MM-DD parser matched this shape. The weekly fallback therefore
|
||||
* fell back to WEEKLY_QUOTA_COOLDOWN_MS (24h) and the connection was dispatched
|
||||
* again — into a real upstream 429 — every day until the true reset ~6 days out.
|
||||
*
|
||||
* The datetime may use a `T` or a space separator, and may carry `Z` or a
|
||||
* `±HH:MM` offset. A NAIVE datetime (no zone) is interpreted as UTC: Z.AI
|
||||
* reports in UTC, and treating it as local time would shift the cooldown by the
|
||||
* host offset. Returns null when the instant is not in the future.
|
||||
*/
|
||||
export function parseIsoDateTimeResetMs(
|
||||
msg: string,
|
||||
maxMs: number,
|
||||
nowMs: number = Date.now()
|
||||
): number | null {
|
||||
const match =
|
||||
/\b(?:try again at|wait until|reset(?:s)?\s+at|available at|retry after)\s+(\d{4}-\d{2}-\d{2}[Tt ]\d{2}:\d{2}(?::\d{2})?(?:\.\d+)?)\s*(Z|[+-]\d{2}:?\d{2})?/i.exec(
|
||||
msg
|
||||
);
|
||||
if (!match) return null;
|
||||
const stamp = match[1].replace(/[Tt ]/, "T");
|
||||
// No zone in the body → UTC (see doc comment). Normalize \"+0200\" to \"+02:00\":
|
||||
// the bare-offset form is not part of the ES Date.parse grammar.
|
||||
const rawZone = match[2] ? match[2].toUpperCase() : "Z";
|
||||
const zone = /^[+-]\d{4}$/.test(rawZone)
|
||||
? `${rawZone.slice(0, 3)}:${rawZone.slice(3)}`
|
||||
: rawZone;
|
||||
const resetMs = Date.parse(`${stamp}${zone}`);
|
||||
if (!Number.isFinite(resetMs)) return null;
|
||||
const waitMs = resetMs - nowMs;
|
||||
if (waitMs <= 0) return null;
|
||||
return Math.min(waitMs, maxMs);
|
||||
}
|
||||
|
||||
/**
|
||||
* Qwen token-plan (and similar apikey providers) report the weekly reset as
|
||||
* \"The quota will reset at 08-29 15:29:00 UTC\" without a year. Treat that as
|
||||
* the next occurrence of MM-DD HH:MM[:SS] UTC; if the date already passed this
|
||||
* year, roll to next year. Returns null when the parsed instant is not in the
|
||||
* future or the wait would exceed maxMs.
|
||||
*/
|
||||
export function parseMonthDayResetMs(
|
||||
msg: string,
|
||||
maxMs: number,
|
||||
nowMs: number = Date.now()
|
||||
): number | null {
|
||||
const match =
|
||||
/reset(?:s)?\s+at\s+(\d{2})-(\d{2})\s+(\d{2}):(\d{2})(?::(\d{2}))?\s*(?:UTC|Z)?/i.exec(
|
||||
msg
|
||||
);
|
||||
if (!match) return null;
|
||||
const month = Number.parseInt(match[1], 10);
|
||||
const day = Number.parseInt(match[2], 10);
|
||||
const hour = Number.parseInt(match[3], 10);
|
||||
const minute = Number.parseInt(match[4], 10);
|
||||
const second = match[5] ? Number.parseInt(match[5], 10) : 0;
|
||||
if (
|
||||
month < 1 ||
|
||||
month > 12 ||
|
||||
day < 1 ||
|
||||
day > 31 ||
|
||||
hour > 23 ||
|
||||
minute > 59 ||
|
||||
second > 59
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
const now = new Date(nowMs);
|
||||
let year = now.getUTCFullYear();
|
||||
let resetMs = Date.UTC(year, month - 1, day, hour, minute, second);
|
||||
if (!Number.isFinite(resetMs)) return null;
|
||||
if (resetMs <= nowMs) {
|
||||
year += 1;
|
||||
resetMs = Date.UTC(year, month - 1, day, hour, minute, second);
|
||||
}
|
||||
const waitMs = resetMs - nowMs;
|
||||
if (!Number.isFinite(waitMs) || waitMs <= 0) return null;
|
||||
return Math.min(waitMs, maxMs);
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
*/
|
||||
|
||||
import { RateLimitReason } from "../config/constants.ts";
|
||||
import { parseDayGranularityResetMs } from "./quotaResetParsing.ts";
|
||||
|
||||
type RateLimitReasonValue = (typeof RateLimitReason)[keyof typeof RateLimitReason];
|
||||
|
||||
@@ -97,16 +98,29 @@ export function isWeeklyUsageLimitText(lower: string): boolean {
|
||||
return (
|
||||
lower.includes("weekly usage limit") ||
|
||||
lower.includes("weekly limit reached") ||
|
||||
lower.includes("reached your weekly")
|
||||
lower.includes("reached your weekly") ||
|
||||
lower.includes("1-week quota") ||
|
||||
lower.includes("week quota") ||
|
||||
lower.includes("weekly/monthly limit") ||
|
||||
(lower.includes("weekly") && lower.includes("quota") && lower.includes("exhaust"))
|
||||
);
|
||||
}
|
||||
|
||||
const MAX_WEEKLY_QUOTA_COOLDOWN_MS = 30 * 24 * 60 * 60 * 1000;
|
||||
|
||||
export function buildWeeklyQuotaFallback(errorStr: string): QuotaTextFallback | null {
|
||||
if (!isWeeklyUsageLimitText(errorStr.toLowerCase())) return null;
|
||||
const parsedResetMs = parseDayGranularityResetMs(errorStr, MAX_WEEKLY_QUOTA_COOLDOWN_MS);
|
||||
const cooldownMs =
|
||||
typeof parsedResetMs === "number" && parsedResetMs > 0
|
||||
? parsedResetMs
|
||||
: WEEKLY_QUOTA_COOLDOWN_MS;
|
||||
return {
|
||||
shouldFallback: true,
|
||||
cooldownMs: WEEKLY_QUOTA_COOLDOWN_MS,
|
||||
cooldownMs,
|
||||
reason: RateLimitReason.QUOTA_EXHAUSTED,
|
||||
usedUpstreamRetryHint: typeof parsedResetMs === "number" && parsedResetMs > 0,
|
||||
quotaResetHintMs: typeof parsedResetMs === "number" && parsedResetMs > 0 ? parsedResetMs : undefined,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -64,7 +64,7 @@ const MAX_CONVERSATION_AFFINITY_ENTRIES = 1000;
|
||||
* Task routing is additive: other strategies are wholly unaffected.
|
||||
*/
|
||||
export function isTaskRoutingStrategy(strategy: unknown): boolean {
|
||||
return ["smart", "task", "task-aware", "task_aware", "auto"].includes(
|
||||
return ["smart", "task", "task-aware", "task_aware"].includes(
|
||||
String(strategy ?? "").toLowerCase()
|
||||
);
|
||||
}
|
||||
|
||||